{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-13T04:49:40.198684Z","iopub.execute_input":"2023-09-13T04:49:40.199156Z","iopub.status.idle":"2023-09-13T04:49:40.238097Z","shell.execute_reply.started":"2023-09-13T04:49:40.199122Z","shell.execute_reply":"2023-09-13T04:49:40.237043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**This is my first notebook and I am hoping to improve as time progresses. In this notebook, I will load the training data and reduce the memory usage by 50% . The source for memory reduction technique is  https://www.kaggle.com/code/kyakovlev/ieee-fe-for-local-test. After reducing the memory I will perform brief analysis on the training data to understand it better**","metadata":{}},{"cell_type":"code","source":"#set the directories\nimport os\ndata_dir = r'/kaggle/input/stanford-ribonanza-rna-folding'\nbpp_files_dir = os.path.join(data_dir,'Ribonanza_bpp_files')\neterna_metadata_dir = os.path.join(data_dir,'eterna_openknot_metadata')\nseq_libraries_dir = os.path.join(data_dir,'sequence_libraries')\nsilico_pred_dir = os.path.join(data_dir,'supplementary_silico_predictions')\nsample_submission_file = os.path.join(data_dir,'sample_submission.csv')\ntest_seq_file = os.path.join(data_dir,'test_sequences.csv')\ntrain_data_file = os.path.join(data_dir,'train_data.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-13T05:42:19.897361Z","iopub.execute_input":"2023-09-13T05:42:19.897771Z","iopub.status.idle":"2023-09-13T05:42:19.905667Z","shell.execute_reply.started":"2023-09-13T05:42:19.897739Z","shell.execute_reply":"2023-09-13T05:42:19.904397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib as plt","metadata":{"execution":{"iopub.status.busy":"2023-09-13T05:42:23.406683Z","iopub.execute_input":"2023-09-13T05:42:23.407225Z","iopub.status.idle":"2023-09-13T05:42:24.446041Z","shell.execute_reply.started":"2023-09-13T05:42:23.407177Z","shell.execute_reply":"2023-09-13T05:42:24.444429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(train_data_file)","metadata":{"execution":{"iopub.status.busy":"2023-09-13T05:42:59.070985Z","iopub.execute_input":"2023-09-13T05:42:59.071704Z","iopub.status.idle":"2023-09-13T05:45:01.534258Z","shell.execute_reply.started":"2023-09-13T05:42:59.071663Z","shell.execute_reply":"2023-09-13T05:45:01.533096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2023-09-13T05:45:22.970781Z","iopub.execute_input":"2023-09-13T05:45:22.971201Z","iopub.status.idle":"2023-09-13T05:45:23.027956Z","shell.execute_reply.started":"2023-09-13T05:45:22.971168Z","shell.execute_reply":"2023-09-13T05:45:23.026236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Quick analysis on the dataset:**<br>\n*There are 1.65 million records, with 419 columns each*<br>\n*It requires more than 5GB of memory to store the data*","metadata":{}},{"cell_type":"code","source":"#Helper function to reduce memory usage\n#source: https://www.kaggle.com/code/kyakovlev/ieee-fe-for-local-test\ndef reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"execution":{"iopub.status.busy":"2023-09-13T05:45:35.481418Z","iopub.execute_input":"2023-09-13T05:45:35.481849Z","iopub.status.idle":"2023-09-13T05:45:35.497673Z","shell.execute_reply.started":"2023-09-13T05:45:35.481817Z","shell.execute_reply":"2023-09-13T05:45:35.496286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reduce memory usage by assigning the real datatypes based on the data\ntrain_data = reduce_mem_usage(train_data)","metadata":{"execution":{"iopub.status.busy":"2023-09-13T05:45:49.919804Z","iopub.execute_input":"2023-09-13T05:45:49.920417Z","iopub.status.idle":"2023-09-13T05:46:01.914488Z","shell.execute_reply.started":"2023-09-13T05:45:49.920365Z","shell.execute_reply":"2023-09-13T05:46:01.913121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-09-13T05:51:13.805950Z","iopub.execute_input":"2023-09-13T05:51:13.807257Z","iopub.status.idle":"2023-09-13T05:51:13.851097Z","shell.execute_reply.started":"2023-09-13T05:51:13.807199Z","shell.execute_reply":"2023-09-13T05:51:13.849647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Things to check** <br>\nHow many unique sequences available? <br>\nDoes training data has DNA sequence in it? RNA has combination of only (GACU) while DNA has (GACT) <br>\nHow many experiment types in total? <br>\nHow many experiment types per sequence? <br>\nHow many with SN_filter>0 ? <br>\n","metadata":{}},{"cell_type":"code","source":"# How many unique sequences available?\nprint(\"Total unique RNA sequences: {:,}\".format(len(train_data['sequence'].unique())))","metadata":{"execution":{"iopub.status.busy":"2023-09-13T05:46:26.574013Z","iopub.execute_input":"2023-09-13T05:46:26.574515Z","iopub.status.idle":"2023-09-13T05:46:27.696079Z","shell.execute_reply.started":"2023-09-13T05:46:26.574476Z","shell.execute_reply":"2023-09-13T05:46:27.694700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**There are 806,573 unique RNA sequences available in the dataset**","metadata":{}},{"cell_type":"code","source":"#Are there any DNA sequences?\nDNA_data = train_data.loc[train_data['sequence'].str.contains(\"T\", case=False)]","metadata":{"execution":{"iopub.status.busy":"2023-09-13T06:17:39.100491Z","iopub.execute_input":"2023-09-13T06:17:39.100977Z","iopub.status.idle":"2023-09-13T06:17:44.170194Z","shell.execute_reply.started":"2023-09-13T06:17:39.100944Z","shell.execute_reply":"2023-09-13T06:17:44.168846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total DNA sequences in training data:  {}\".format(DNA_data.shape[0]))","metadata":{"execution":{"iopub.status.busy":"2023-09-13T06:18:52.175973Z","iopub.execute_input":"2023-09-13T06:18:52.176452Z","iopub.status.idle":"2023-09-13T06:18:52.183109Z","shell.execute_reply.started":"2023-09-13T06:18:52.176414Z","shell.execute_reply":"2023-09-13T06:18:52.181814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#How many observations available per experiment type?\ntrain_data['experiment_type'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-09-13T06:21:40.598292Z","iopub.execute_input":"2023-09-13T06:21:40.598873Z","iopub.status.idle":"2023-09-13T06:21:40.892464Z","shell.execute_reply.started":"2023-09-13T06:21:40.598836Z","shell.execute_reply":"2023-09-13T06:21:40.891040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**The dataset is equally divided between two experiments. However, number of observations per experiment is greater than the unique RNA sequences. What does this say?**\n\n\n**Assuming each RNA sequence go through two experiments** \n-> Total observations should be (806,573 * 2) = 1,613,146. However, total observations in the training dataset =  1,643,680\n\n**Delta = 1,643,680 - 1,613,146 = 30,534**\n\n\nWe need to investigate **what are the additional 30,534 observations represent?** ","metadata":{}},{"cell_type":"code","source":"#save reduced dataset\ntrain_data.to_csv(os.path.join('/kaggle/working','reduced_training_data.csv'))","metadata":{"execution":{"iopub.status.busy":"2023-09-13T06:27:00.388984Z","iopub.execute_input":"2023-09-13T06:27:00.390167Z","iopub.status.idle":"2023-09-13T06:38:12.181961Z","shell.execute_reply.started":"2023-09-13T06:27:00.390116Z","shell.execute_reply":"2023-09-13T06:38:12.180442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}