{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## The dataset is [here](https://www.kaggle.com/datasets/horikitasaku/rna-low-m-data)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\n\n#Ignore warnings\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:30:33.044682Z","iopub.execute_input":"2023-09-11T01:30:33.045022Z","iopub.status.idle":"2023-09-11T01:30:33.383499Z","shell.execute_reply.started":"2023-09-11T01:30:33.044993Z","shell.execute_reply":"2023-09-11T01:30:33.382503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/train_data.csv')\ntrain_df.head()\n# CPU times: user 1min 5s, sys: 14.9 s, total: 1min 20s\n# Wall time: 1min 37s","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:04:14.532160Z","iopub.execute_input":"2023-09-11T01:04:14.532590Z","iopub.status.idle":"2023-09-11T01:05:53.865040Z","shell.execute_reply.started":"2023-09-11T01:04:14.532561Z","shell.execute_reply":"2023-09-11T01:05:53.863984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Bytes before dropped:',train_df.memory_usage(deep=True).sum())","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:05:53.867012Z","iopub.execute_input":"2023-09-11T01:05:53.867637Z","iopub.status.idle":"2023-09-11T01:05:54.716078Z","shell.execute_reply.started":"2023-09-11T01:05:53.867604Z","shell.execute_reply":"2023-09-11T01:05:54.714987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('original train data shape:',train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:05:54.718453Z","iopub.execute_input":"2023-09-11T01:05:54.718752Z","iopub.status.idle":"2023-09-11T01:05:54.723124Z","shell.execute_reply.started":"2023-09-11T01:05:54.718725Z","shell.execute_reply":"2023-09-11T01:05:54.722023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"```python\ntrain_df.reactivity_0001.unique(),train_df.reactivity_0002.unique()\n(array([nan]), array([nan]))\n```\n## As we can see, lots of the reactivity_* is nan, we can simply drop all nan columns.\n## All 'reactivity_*' respond to a certain base, then all nan columns have no meaning, can be directly dropped.","metadata":{}},{"cell_type":"code","source":"train_df = train_df.dropna(axis=1,how='all')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:05:54.724366Z","iopub.execute_input":"2023-09-11T01:05:54.724669Z","iopub.status.idle":"2023-09-11T01:05:57.439774Z","shell.execute_reply.started":"2023-09-11T01:05:54.724641Z","shell.execute_reply":"2023-09-11T01:05:57.438703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Bytes after dropped:',train_df.memory_usage(deep=True).sum())","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:05:57.441098Z","iopub.execute_input":"2023-09-11T01:05:57.441397Z","iopub.status.idle":"2023-09-11T01:05:58.257430Z","shell.execute_reply.started":"2023-09-11T01:05:57.441372Z","shell.execute_reply":"2023-09-11T01:05:58.256491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('valid train data shape:',train_df.shape)","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:05:58.258534Z","iopub.execute_input":"2023-09-11T01:05:58.258908Z","iopub.status.idle":"2023-09-11T01:05:58.263198Z","shell.execute_reply.started":"2023-09-11T01:05:58.258882Z","shell.execute_reply":"2023-09-11T01:05:58.262517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_parquet('train_data.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:05:58.264097Z","iopub.execute_input":"2023-09-11T01:05:58.264788Z","iopub.status.idle":"2023-09-11T01:06:18.684350Z","shell.execute_reply.started":"2023-09-11T01:05:58.264761Z","shell.execute_reply":"2023-09-11T01:06:18.683020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = pd.read_parquet('/kaggle/working/train_data.parquet')\ntrain_df.head()\n# CPU times: user 9.19 s, sys: 7.49 s, total: 16.7 s\n# Wall time: 5.52 s","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:09:34.555593Z","iopub.execute_input":"2023-09-11T01:09:34.556044Z","iopub.status.idle":"2023-09-11T01:09:40.107893Z","shell.execute_reply.started":"2023-09-11T01:09:34.556014Z","shell.execute_reply":"2023-09-11T01:09:40.107136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The effect of drop and converting is very obvious.\n\n|                | Before     | After      | Reduced       |\n| -------------- | ---------- | ---------- | ------------- |\n| columns        | 419        | 285        | 134           |\n| memory (bytes) | 6240692318 | 4478667358 | 1,762,024,960 |\n| Wall time      | 1min 39s   | 5.52 s     | 1min 33s      |\n\nDo the same thing to test_sequences.csv and sample_submission.csv","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/sample_submission.csv')\nsubmission.to_parquet('submission.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:26:55.247278Z","iopub.execute_input":"2023-09-11T01:26:55.247779Z","iopub.status.idle":"2023-09-11T01:29:01.838528Z","shell.execute_reply.started":"2023-09-11T01:26:55.247747Z","shell.execute_reply":"2023-09-11T01:29:01.836495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsubmission = pd.read_parquet('submission.parquet')\n# CPU times: user 10.1 s, sys: 23.5 s, total: 33.6 s\n# Wall time: 13.9 s","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:29:24.163233Z","iopub.execute_input":"2023-09-11T01:29:24.163696Z","iopub.status.idle":"2023-09-11T01:29:38.087167Z","shell.execute_reply.started":"2023-09-11T01:29:24.163649Z","shell.execute_reply":"2023-09-11T01:29:38.086169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/stanford-ribonanza-rna-folding/test_sequences.csv')\ntest.to_parquet('test_sequences.parquet')\n","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:30:33.385010Z","iopub.execute_input":"2023-09-11T01:30:33.385406Z","iopub.status.idle":"2023-09-11T01:30:42.172884Z","shell.execute_reply.started":"2023-09-11T01:30:33.385356Z","shell.execute_reply":"2023-09-11T01:30:42.171929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nsubmission = pd.read_parquet('test_sequences.parquet')\n# CPU times: user 1.19 s, sys: 636 ms, total: 1.83 s\n# Wall time: 1.67 s","metadata":{"execution":{"iopub.status.busy":"2023-09-11T01:30:57.532869Z","iopub.execute_input":"2023-09-11T01:30:57.533218Z","iopub.status.idle":"2023-09-11T01:30:59.208325Z","shell.execute_reply.started":"2023-09-11T01:30:57.533189Z","shell.execute_reply":"2023-09-11T01:30:59.206730Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}