{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":51294,"databundleVersionId":6923401,"sourceType":"competition"},{"sourceId":6822004,"sourceType":"datasetVersion","datasetId":3719560},{"sourceId":6869273,"sourceType":"datasetVersion","datasetId":3947568},{"sourceId":6869280,"sourceType":"datasetVersion","datasetId":3947574},{"sourceId":6881390,"sourceType":"datasetVersion","datasetId":3949409},{"sourceId":6903935,"sourceType":"datasetVersion","datasetId":3965347},{"sourceId":6915484,"sourceType":"datasetVersion","datasetId":3971075},{"sourceId":6969973,"sourceType":"datasetVersion","datasetId":3999308},{"sourceId":6970774,"sourceType":"datasetVersion","datasetId":3949435},{"sourceId":6982107,"sourceType":"datasetVersion","datasetId":4012608}],"dockerImageVersionId":30558,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Visualize the location distribution of outliers and 0 values in a sequence.\n\n#### Just like the [discussion](https://www.kaggle.com/competitions/stanford-ribonanza-rna-folding/discussion/444653) said, the model will output some value that shouldn't occur, such as a completely 0, so I want to see the positions in the sequence most of these outliers are distributed.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# r2 = pd.read_csv('/kaggle/input/v7-0sub/submission.csv')\nr1 = pd.read_csv('/kaggle/input/v1/submission.csv') ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-16T17:42:15.135462Z","iopub.execute_input":"2023-11-16T17:42:15.136010Z","iopub.status.idle":"2023-11-16T17:48:21.913079Z","shell.execute_reply.started":"2023-11-16T17:42:15.135961Z","shell.execute_reply":"2023-11-16T17:48:21.910807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Since the data I was testing with didn't have a zero value, I used < 0.001, you can simply set ==0 to see the 0 value\n\n#### In addition, I think that a perfect 1 can also be abnormal, so maybe it is necessary to check...","metadata":{}},{"cell_type":"code","source":"zero_indices_2A3 = r1[r1['reactivity_2A3_MaP'] < 0.001].index\nzero_indices_DMS = r1[r1['reactivity_DMS_MaP'] < 0.001].index\n\none_indices_2A3 = r1[r1['reactivity_2A3_MaP'] > 0.99].index\none_indices_DMS = r1[r1['reactivity_DMS_MaP'] > 0.99].index","metadata":{"execution":{"iopub.status.busy":"2023-11-16T17:48:36.834224Z","iopub.execute_input":"2023-11-16T17:48:36.834681Z","iopub.status.idle":"2023-11-16T17:48:38.277674Z","shell.execute_reply.started":"2023-11-16T17:48:36.834646Z","shell.execute_reply":"2023-11-16T17:48:38.276281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r1.loc[zero_indices_2A3, 'reactivity_2A3_MaP']","metadata":{"execution":{"iopub.status.busy":"2023-11-16T17:48:23.650685Z","iopub.execute_input":"2023-11-16T17:48:23.651153Z","iopub.status.idle":"2023-11-16T17:48:23.673384Z","shell.execute_reply.started":"2023-11-16T17:48:23.651116Z","shell.execute_reply":"2023-11-16T17:48:23.672460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_parquet('/kaggle/input/stanford-ribonanza-rna-folding-converted/test_sequences.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-11-16T17:48:23.675515Z","iopub.execute_input":"2023-11-16T17:48:23.676226Z","iopub.status.idle":"2023-11-16T17:48:28.022117Z","shell.execute_reply.started":"2023-11-16T17:48:23.676186Z","shell.execute_reply":"2023-11-16T17:48:28.020573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[['id_min','id_max']]","metadata":{"execution":{"iopub.status.busy":"2023-11-16T17:48:28.023544Z","iopub.execute_input":"2023-11-16T17:48:28.024105Z","iopub.status.idle":"2023-11-16T17:48:28.061216Z","shell.execute_reply.started":"2023-11-16T17:48:28.024056Z","shell.execute_reply":"2023-11-16T17:48:28.059812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2A3 reactions near 0 and 1.","metadata":{}},{"cell_type":"code","source":"from tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\ntest_df['length'] = test_df['id_max'] - test_df['id_min'] + 1\n\nindex_df = pd.DataFrame({'index': zero_indices_DMS})\n\nmatched_df = pd.merge_asof(index_df.sort_values('index'), test_df.sort_values('id_min'), left_on='index', right_on='id_min')\n\nmatched_df['position'] = (matched_df['index'] - matched_df['id_min']) / matched_df['length']\n\nplt.hist(matched_df['position'], bins=100, range=(0,1))\nplt.xlabel('Nearly zero_indices_DMS Position in Sequence')\nplt.ylabel('Count')\nplt.title('Distribution of Index Positions in Sequences')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-16T17:48:28.063093Z","iopub.execute_input":"2023-11-16T17:48:28.064441Z","iopub.status.idle":"2023-11-16T17:48:29.532624Z","shell.execute_reply.started":"2023-11-16T17:48:28.064390Z","shell.execute_reply":"2023-11-16T17:48:29.531185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index_df = pd.DataFrame({'index': one_indices_2A3})\n\nmatched_df = pd.merge_asof(index_df.sort_values('index'), test_df.sort_values('id_min'), left_on='index', right_on='id_min')\n\nmatched_df['position'] = (matched_df['index'] - matched_df['id_min']) / matched_df['length']\n\nplt.hist(matched_df['position'], bins=100, range=(0,1))\nplt.xlabel('Nearly one_indices_2A3 Position in Sequence')\nplt.ylabel('Count')\nplt.title('Distribution of Index Positions in Sequences')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-16T17:48:41.014840Z","iopub.execute_input":"2023-11-16T17:48:41.015294Z","iopub.status.idle":"2023-11-16T17:48:41.721407Z","shell.execute_reply.started":"2023-11-16T17:48:41.015259Z","shell.execute_reply":"2023-11-16T17:48:41.719989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DMS reactions near 0 and 1.","metadata":{}},{"cell_type":"code","source":"index_df = pd.DataFrame({'index': zero_indices_2A3})\n\nmatched_df = pd.merge_asof(index_df.sort_values('index'), test_df.sort_values('id_min'), left_on='index', right_on='id_min')\n\nmatched_df['position'] = (matched_df['index'] - matched_df['id_min']) / matched_df['length']\n\nplt.hist(matched_df['position'], bins=100, range=(0,1))\nplt.xlabel('Nearly zero_indices_2A3 Position in Sequence')\nplt.ylabel('Count')\nplt.title('Distribution of Index Positions in Sequences')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-16T17:48:29.534594Z","iopub.execute_input":"2023-11-16T17:48:29.535057Z","iopub.status.idle":"2023-11-16T17:48:30.233037Z","shell.execute_reply.started":"2023-11-16T17:48:29.535018Z","shell.execute_reply":"2023-11-16T17:48:30.231942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index_df = pd.DataFrame({'index': one_indices_DMS})\n\nmatched_df = pd.merge_asof(index_df.sort_values('index'), test_df.sort_values('id_min'), left_on='index', right_on='id_min')\n\nmatched_df['position'] = (matched_df['index'] - matched_df['id_min']) / matched_df['length']\n\nplt.hist(matched_df['position'], bins=100, range=(0,1))\nplt.xlabel('Nearly one_indices_DMS Position in Sequence')\nplt.ylabel('Count')\nplt.title('Distribution of Index Positions in Sequences')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-16T17:48:44.999912Z","iopub.execute_input":"2023-11-16T17:48:45.001524Z","iopub.status.idle":"2023-11-16T17:48:45.694408Z","shell.execute_reply.started":"2023-11-16T17:48:45.001462Z","shell.execute_reply":"2023-11-16T17:48:45.693164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## This is actually more than I expected, because I guess the heads and tails of the sequence are the positions that have the most outliers?","metadata":{}}]}