{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nHere we look at what non-standard positions we have nonNAN targets.\n\nBy standard - we mean 100 (!) positions, omiting the first 26.  These are non-NANs for ALL train data. I.e. data[26:126] - non-nans  \n\nVersion 1 - we look only on selected subsamples of 2A3 target  - stored and prepared before: https://www.kaggle.com/code/alexandervc/ribonanza-prepare-data\n\nWe get the following groups:\n\n    ### Positions 126-129 here 16033 non-nans - mainly ( 13627 ): length 170, database 15k_2A3 , some others \n    ### Positions 135-165 - only 2384 nonNANs - all length 206, database: SL5_M2seq_2A3\n    ### Positions 129-135 - only 2406 - non-NAN positions - same as above +    22 rna of length 155\n    \n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time() \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\ncc = 0\nfor dirname, _, filenames in os.walk('/kaggle/input/stanford-ribonanza-rna-folding-converted/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\nfor dirname, _, filenames in os.walk('/kaggle/input/stanford-ribonanza-rna-folding-data/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    if 'stanford-ribonanza-rna-folding-converted' in dirname: continue \n    if 'stanford-ribonanza-rna-folding-data' in dirname: continue \n    for filename in filenames:\n        cc += 1\n        if cc < 20:\n            print(os.path.join(dirname, filename))\n        else:\n            break\n    if cc < 20:\n        pass\n    else:\n        break\n\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-24T08:11:54.214801Z","iopub.execute_input":"2023-09-24T08:11:54.215567Z","iopub.status.idle":"2023-09-24T08:11:56.946150Z","shell.execute_reply.started":"2023-09-24T08:11:54.215527Z","shell.execute_reply":"2023-09-24T08:11:56.944719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load sequences, SNR and other technical part of train data","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/stanford-ribonanza-rna-folding-data/cleaned_163428_rna/train_cleaned_2A3_MaP_163428_info_only.csv'\ndf_2A3_inf = pd.read_csv(fn, index_col = 0)\nprint(df_2A3_inf.shape)\ndf_2A3_inf\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T08:13:07.812389Z","iopub.execute_input":"2023-09-24T08:13:07.814664Z","iopub.status.idle":"2023-09-24T08:13:09.160517Z","shell.execute_reply.started":"2023-09-24T08:13:07.814603Z","shell.execute_reply":"2023-09-24T08:13:09.159259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/stanford-ribonanza-rna-folding-data/cleaned_163428_rna/train_cleaned_DMS_MaP_163428_info_only.csv'\ndf_DMS_inf = pd.read_csv(fn, index_col = 0)\nprint(df_DMS_inf.shape)\ndf_DMS_inf\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T08:13:38.858888Z","iopub.execute_input":"2023-09-24T08:13:38.859329Z","iopub.status.idle":"2023-09-24T08:13:40.114758Z","shell.execute_reply.started":"2023-09-24T08:13:38.859296Z","shell.execute_reply":"2023-09-24T08:13:40.113590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Targets related - reactivity and reactivity errors","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/stanford-ribonanza-rna-folding-data/cleaned_163428_rna/train_cleaned_2A3_MaP_163428_reactivity.npy'\nY_2A3_reactivity = np.load(fn)\nprint(Y_2A3_reactivity.shape)\n\nfn = '/kaggle/input/stanford-ribonanza-rna-folding-data/cleaned_163428_rna/train_cleaned_DMS_MaP_163428_reactivity.npy'\nY_DMS_reactivity = np.load(fn)\nprint(Y_DMS_reactivity.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T08:14:06.890572Z","iopub.execute_input":"2023-09-24T08:14:06.891564Z","iopub.status.idle":"2023-09-24T08:14:13.056050Z","shell.execute_reply.started":"2023-09-24T08:14:06.891530Z","shell.execute_reply":"2023-09-24T08:14:13.054855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Look on NANs\n\n    Postions v[26:126] - no NANs at all \n    ### Position 126-129 here 16033 non-nans - mainly ( 13627 ): length 170, database 15k_2A3 , some others \n    ### positions 129-135 - only 2406 - non-NAN positions \n    ### Positions 135-165 - only 2384 nonNANs - all length 206, database: SL5_M2seq_2A3\n    First 26 - all are NANs\n    ","metadata":{}},{"cell_type":"code","source":"%%time\nprint(Y_2A3_reactivity.shape)\n\nv = np.isnan(Y_2A3_reactivity).sum(axis=0)\nprint(len(v))\nprint(list(v))\nprint(v[25],v[26], v[125:127], v[200])\nfor k in range(len(v)-1):\n    if (v[k] == 0) and (v[k+1]!=0): print(k)\n    if (v[k+1] == 0) and (v[k]!=0): print(k)\nprint(list(v[25:127]))        \nprint(); print()\n\nv = np.isnan(Y_DMS_reactivity).sum(axis=0)\nprint(len(v))\nprint(list(v))\nprint(v[25],v[26], v[125:127], v[200])\nfor k in range(len(v)-1):\n    if (v[k] == 0) and (v[k+1]!=0): print(k)\n    if (v[k+1] == 0) and (v[k]!=0): print(k)\nprint(list(v[25:127]))        \nprint(); print()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T08:14:53.814483Z","iopub.execute_input":"2023-09-24T08:14:53.814937Z","iopub.status.idle":"2023-09-24T08:14:54.006136Z","shell.execute_reply.started":"2023-09-24T08:14:53.814903Z","shell.execute_reply":"2023-09-24T08:14:54.005001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v[125:137], v[137:147], v[147:157], v[157:167], v[164:166]","metadata":{"execution":{"iopub.status.busy":"2023-09-24T08:20:48.104916Z","iopub.execute_input":"2023-09-24T08:20:48.105344Z","iopub.status.idle":"2023-09-24T08:20:48.114722Z","shell.execute_reply.started":"2023-09-24T08:20:48.105313Z","shell.execute_reply":"2023-09-24T08:20:48.113495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_2A3_inf.shape)\nfor k in range(125,170):\n    if v[k-1] != v[k]:\n        print(k,  v[k-1],v[k])","metadata":{"execution":{"iopub.status.busy":"2023-09-24T08:43:28.274804Z","iopub.execute_input":"2023-09-24T08:43:28.275162Z","iopub.status.idle":"2023-09-24T08:43:28.282215Z","shell.execute_reply.started":"2023-09-24T08:43:28.275131Z","shell.execute_reply":"2023-09-24T08:43:28.280885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Position 126-129 here 16033 non-nans - mainly ( 13627 ): length 170, database 15k_2A3 , some others ","metadata":{}},{"cell_type":"code","source":"print( 163428 - v[126],163428- v[127],163428-v[128] )\nprint()\ndisplay( df_2A3_inf.head(1) )\nprint()\n\nm = (np.isnan(Y_2A3_reactivity[:,126]) == False)\nprint( m.sum() )\nl = [len(t) for t in df_2A3_inf.iloc[m,:]['sequence'] ]\nprint( pd.Series(l).value_counts() )\nprint(); print();\n\nl = [ t for t in df_2A3_inf.iloc[m,:]['dataset_name'] ]\nprint( pd.Series(l).value_counts() )\n\ndf_2A3_inf['dataset_name'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-09-24T08:28:42.350103Z","iopub.execute_input":"2023-09-24T08:28:42.350603Z","iopub.status.idle":"2023-09-24T08:28:42.422525Z","shell.execute_reply.started":"2023-09-24T08:28:42.350563Z","shell.execute_reply":"2023-09-24T08:28:42.421239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l = [len(t) for t in df_2A3_inf['sequence'] ]\npd.Series(l).value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-09-24T08:47:48.562433Z","iopub.execute_input":"2023-09-24T08:47:48.562835Z","iopub.status.idle":"2023-09-24T08:47:48.705437Z","shell.execute_reply.started":"2023-09-24T08:47:48.562804Z","shell.execute_reply":"2023-09-24T08:47:48.704245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### positions 129-135 - only 2406 - non-NAN positions, lengths 206,155","metadata":{}},{"cell_type":"code","source":"print( 163428 - v[129],163428- v[130],163428-v[131] )\nprint()\ndisplay( df_2A3_inf.head(1) )\nprint()\n\nm = (np.isnan(Y_2A3_reactivity[:,129]) == False)\nprint( m.sum() )\nl = [len(t) for t in df_2A3_inf.iloc[m,:]['sequence'] ]\nprint( pd.Series(l).value_counts() )\nprint(); print();\n\nl = [ t for t in df_2A3_inf.iloc[m,:]['dataset_name'] ]\nprint( pd.Series(l).value_counts() )\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T08:30:48.271839Z","iopub.execute_input":"2023-09-24T08:30:48.272321Z","iopub.status.idle":"2023-09-24T08:30:48.299934Z","shell.execute_reply.started":"2023-09-24T08:30:48.272284Z","shell.execute_reply":"2023-09-24T08:30:48.298738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Positions 135-165 - only 2384 nonNANs - all length 206, database: SL5_M2seq_2A3","metadata":{}},{"cell_type":"code","source":"K,K2,K3 =  135,164,165\nprint('K=', K,K2,K3)\nprint( 'count not nan at that position:', 163428 - v[K],163428 - v[K2], 163428 - v[K3] )\nprint()\ndisplay( df_2A3_inf.head(1) )\nprint()\n\nm = (np.isnan(Y_2A3_reactivity[:,K]) == False)\nprint( m.sum() )\nl = [len(t) for t in df_2A3_inf.iloc[m,:]['sequence'] ]\nprint( pd.Series(l).value_counts() )\nprint(); print();\n\nl = [ t for t in df_2A3_inf.iloc[m,:]['dataset_name'] ]\nprint( pd.Series(l).value_counts() )\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-24T08:39:41.409788Z","iopub.execute_input":"2023-09-24T08:39:41.410231Z","iopub.status.idle":"2023-09-24T08:39:41.439831Z","shell.execute_reply.started":"2023-09-24T08:39:41.410199Z","shell.execute_reply":"2023-09-24T08:39:41.438616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )\nprint('%.1f minutes passed total '%( (time.time()-t0start)/60)  )\nprint('%.2f hours passed total '%( (time.time()-t0start)/3600)  )","metadata":{},"execution_count":null,"outputs":[]}]}