{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Summary\nAfter calculation of multiome gene values, we have to select the genes specified in submission_ids.csv. I compare three methods.\n## Result\nProcessing time varies from about 1 minutes to 9 minutes. The fasteest was Method 1. below if you include data type conversion.\n","metadata":{"papermill":{"duration":0.009239,"end_time":"2022-09-06T13:04:31.838408","exception":false,"start_time":"2022-09-06T13:04:31.829169","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os, gc, pickle\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom colorama import Fore, Back, Style\nfrom matplotlib.ticker import MaxNLocator\n\nfrom sklearn.base import BaseEstimator, TransformerMixin\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler, scale\nfrom sklearn.decomposition import PCA\nfrom sklearn.dummy import DummyRegressor\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.linear_model import Ridge, LinearRegression\nfrom sklearn.metrics import mean_squared_error\n\nfrom scipy.sparse import *\nfrom scipy.sparse import linalg\nfrom sklearn.decomposition import TruncatedSVD\nimport random as rd\n\nimport psutil\n\ntry:\n    os.environ['SATURN_IMAGE']\n    kernel = \"saturn_cloud\"\nexcept KeyError:\n    kernel = \"kaggle\"\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nif kernel==\"saturn_cloud\":\n    SPARSE_DIR = \"./\"\n    META_DIR = \"./\"\nelse: # kaggle\n    SPARSE_DIR = \"../input/opmsci-sparse-data/\"\n    META_DIR = \"../input/open-problems-multimodal-singlecell-metadata/\"\n\nMETA_DIR = \"../input/open-problems-multimodal-singlecell-metadata/\"    \nMETA_CITE_TRAIN = os.path.join(META_DIR,\"cite_train_meta.csv\")\nMETA_CITE_TEST = os.path.join(META_DIR,\"cite_test_meta.csv\")\nMETA_MULTI_TRAIN = os.path.join(META_DIR,\"multiome_train_meta.csv\")\nMETA_MULTI_TEST = os.path.join(META_DIR,\"multiome_test_meta.csv\")\n\nSPARSE_DIR = \"../input/opmsci-sparse-data/\"\nSPS_CITE_TRAIN_INPUTS = os.path.join(SPARSE_DIR,\"train_cite_inputs.npz\")\nSPS_CITE_TRAIN_TARGETS = os.path.join(SPARSE_DIR,\"train_cite_targets.npz\")\nSPS_CITE_TEST_INPUTS = os.path.join(SPARSE_DIR,\"test_cite_inputs.npz\")\n\nSPS_MULTIOME_TRAIN_INPUTS = os.path.join(SPARSE_DIR,\"train_multi_inputs.npz\")\nSPS_MULTIOME_TRAIN_TARGETS = os.path.join(SPARSE_DIR,\"train_multi_targets.npz\")\nSPS_MULTIOME_TEST_INPUTS = os.path.join(SPARSE_DIR,\"test_multi_inputs.npz\")\n\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.npz\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")","metadata":{"_kg_hide-input":true,"papermill":{"duration":1.396664,"end_time":"2022-09-06T13:04:33.244681","exception":false,"start_time":"2022-09-06T13:04:31.848017","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:05:27.857117Z","iopub.execute_input":"2022-09-09T00:05:27.857654Z","iopub.status.idle":"2022-09-09T00:05:28.916842Z","shell.execute_reply.started":"2022-09-09T00:05:27.857559Z","shell.execute_reply":"2022-09-09T00:05:28.915745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install -q tables","metadata":{"papermill":{"duration":15.115612,"end_time":"2022-09-06T13:04:48.369966","exception":false,"start_time":"2022-09-06T13:04:33.254354","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:05:28.919103Z","iopub.execute_input":"2022-09-09T00:05:28.919910Z","iopub.status.idle":"2022-09-09T00:05:44.190682Z","shell.execute_reply.started":"2022-09-09T00:05:28.919875Z","shell.execute_reply":"2022-09-09T00:05:44.189301Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check the evaluation_ids.csv is orderly filled","metadata":{"papermill":{"duration":0.00951,"end_time":"2022-09-06T13:04:48.390445","exception":false,"start_time":"2022-09-06T13:04:48.380935","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Read evaluation_ids.csv","metadata":{"papermill":{"duration":0.009513,"end_time":"2022-09-06T13:04:48.410082","exception":false,"start_time":"2022-09-06T13:04:48.400569","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\neval_ids = pd.read_csv(FP_EVALUATION_IDS, index_col='row_id')\neval_ids.shape","metadata":{"papermill":{"duration":112.037585,"end_time":"2022-09-06T13:06:40.457510","exception":false,"start_time":"2022-09-06T13:04:48.419925","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:24:51.456200Z","iopub.execute_input":"2022-09-09T00:24:51.456583Z","iopub.status.idle":"2022-09-09T00:26:39.184343Z","shell.execute_reply.started":"2022-09-09T00:24:51.456552Z","shell.execute_reply":"2022-09-09T00:26:39.183244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"See how many genes does each cell have.","metadata":{"papermill":{"duration":0.01198,"end_time":"2022-09-06T13:06:40.482103","exception":false,"start_time":"2022-09-06T13:06:40.470123","status":"completed"},"tags":[]}},{"cell_type":"code","source":"cell_count = eval_ids.groupby('cell_id').count()\ncell_count","metadata":{"papermill":{"duration":11.603581,"end_time":"2022-09-06T13:06:52.098237","exception":false,"start_time":"2022-09-06T13:06:40.494656","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:07:34.832679Z","iopub.execute_input":"2022-09-09T00:07:34.833009Z","iopub.status.idle":"2022-09-09T00:07:46.282447Z","shell.execute_reply.started":"2022-09-09T00:07:34.832979Z","shell.execute_reply":"2022-09-09T00:07:46.281421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_count.value_counts()","metadata":{"papermill":{"duration":0.033265,"end_time":"2022-09-06T13:06:52.142297","exception":false,"start_time":"2022-09-06T13:06:52.109032","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:07:46.284223Z","iopub.execute_input":"2022-09-09T00:07:46.284663Z","iopub.status.idle":"2022-09-09T00:07:46.296496Z","shell.execute_reply.started":"2022-09-09T00:07:46.284621Z","shell.execute_reply":"2022-09-09T00:07:46.295530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All the cells have 140 or 3512 genes.See the first 48663*140 cells are cite sequence. ","metadata":{"papermill":{"duration":0.00985,"end_time":"2022-09-06T13:06:52.162402","exception":false,"start_time":"2022-09-06T13:06:52.152552","status":"completed"},"tags":[]}},{"cell_type":"code","source":"df_cell = pd.read_csv(FP_CELL_METADATA)\ndf_cell_cite = df_cell[df_cell.technology==\"citeseq\"]\ndf_cell_multi = df_cell[df_cell.technology==\"multiome\"]","metadata":{"papermill":{"duration":0.460251,"end_time":"2022-09-06T13:06:52.634297","exception":false,"start_time":"2022-09-06T13:06:52.174046","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:07:46.297826Z","iopub.execute_input":"2022-09-09T00:07:46.298653Z","iopub.status.idle":"2022-09-09T00:07:46.717139Z","shell.execute_reply.started":"2022-09-09T00:07:46.298622Z","shell.execute_reply":"2022-09-09T00:07:46.716144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"possible_cite_id = eval_ids.iloc[0:(cell_count.gene_id==140).sum()*140]['cell_id']\n((cell_count.gene_id==140).sum()*140) == (possible_cite_id.isin(df_cell_cite.cell_id)).sum()","metadata":{"papermill":{"duration":0.382352,"end_time":"2022-09-06T13:06:53.027116","exception":false,"start_time":"2022-09-06T13:06:52.644764","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:07:46.718309Z","iopub.execute_input":"2022-09-09T00:07:46.718638Z","iopub.status.idle":"2022-09-09T00:07:47.101866Z","shell.execute_reply.started":"2022-09-09T00:07:46.718608Z","shell.execute_reply":"2022-09-09T00:07:47.100488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The first 48663*140 cells are all cite sequence. See that all the 140 genes of the same cell_id are adjacent.","metadata":{"papermill":{"duration":0.009959,"end_time":"2022-09-06T13:06:53.047583","exception":false,"start_time":"2022-09-06T13:06:53.037624","status":"completed"},"tags":[]}},{"cell_type":"code","source":"notallsame = 0\nfor id in range(0,48663*140,140):\n    ids = eval_ids.iloc[id:id+140]['cell_id']\n    if len(ids.unique()) >1:\n        notallsame += 1\n        print(ids)\nnotallsame","metadata":{"papermill":{"duration":7.755503,"end_time":"2022-09-06T13:07:00.813437","exception":false,"start_time":"2022-09-06T13:06:53.057934","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:07:47.103458Z","iopub.execute_input":"2022-09-09T00:07:47.104125Z","iopub.status.idle":"2022-09-09T00:07:54.650569Z","shell.execute_reply.started":"2022-09-09T00:07:47.104081Z","shell.execute_reply":"2022-09-09T00:07:54.649706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All 140 genes belonging to the same cell_id are adjacent.\nSee the order of gene_id in one cell_id is always the same.","metadata":{"papermill":{"duration":0.010595,"end_time":"2022-09-06T13:07:00.834642","exception":false,"start_time":"2022-09-06T13:07:00.824047","status":"completed"},"tags":[]}},{"cell_type":"code","source":"notallsame = 0\nfor id in range(0,140):\n    ids = eval_ids.iloc[id:48663*140:140]['gene_id']\n    if len(ids.unique()) >1:\n        notallsame += 1\n        print(ids)\nnotallsame","metadata":{"papermill":{"duration":1.32707,"end_time":"2022-09-06T13:07:02.173558","exception":false,"start_time":"2022-09-06T13:07:00.846488","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:07:54.651869Z","iopub.execute_input":"2022-09-09T00:07:54.652905Z","iopub.status.idle":"2022-09-09T00:07:55.503396Z","shell.execute_reply.started":"2022-09-09T00:07:54.652870Z","shell.execute_reply":"2022-09-09T00:07:55.502649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The order of gene_id in one cell_id is always the same.Check the order of cite test and evaluation are the same.","metadata":{"papermill":{"duration":0.010141,"end_time":"2022-09-06T13:07:02.194383","exception":false,"start_time":"2022-09-06T13:07:02.184242","status":"completed"},"tags":[]}},{"cell_type":"code","source":"cite_test_hdf = pd.read_hdf(FP_CITE_TEST_INPUTS)","metadata":{"papermill":{"duration":39.099438,"end_time":"2022-09-06T13:07:41.304402","exception":false,"start_time":"2022-09-06T13:07:02.204964","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:07:55.506872Z","iopub.execute_input":"2022-09-09T00:07:55.507229Z","iopub.status.idle":"2022-09-09T00:08:34.893857Z","shell.execute_reply.started":"2022-09-09T00:07:55.507197Z","shell.execute_reply":"2022-09-09T00:08:34.892916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"notsame = 0\ncite_test_hdf.index[0]\nfor idx in range(0,48663):\n    if eval_ids.iloc[idx*140]['cell_id'] != cite_test_hdf.index[idx]:\n        notsame += 1\nnotsame","metadata":{"papermill":{"duration":3.499371,"end_time":"2022-09-06T13:07:44.815308","exception":false,"start_time":"2022-09-06T13:07:41.315937","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:08:34.895287Z","iopub.execute_input":"2022-09-09T00:08:34.895827Z","iopub.status.idle":"2022-09-09T00:08:38.311684Z","shell.execute_reply.started":"2022-09-09T00:08:34.895796Z","shell.execute_reply":"2022-09-09T00:08:38.310560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" OK for cite sequence. <br> Check for multiome","metadata":{"papermill":{"duration":0.01084,"end_time":"2022-09-06T13:07:44.837351","exception":false,"start_time":"2022-09-06T13:07:44.826511","status":"completed"},"tags":[]}},{"cell_type":"code","source":"notallsame = 0\nfor id in range(48663*140,65744180,3512):\n    ids = eval_ids.iloc[id:id+3512]['cell_id']\n    if len(ids.unique()) >1:\n        notallsame += 1\nnotallsame","metadata":{"papermill":{"duration":7.182703,"end_time":"2022-09-06T13:07:52.031356","exception":false,"start_time":"2022-09-06T13:07:44.848653","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:08:38.313574Z","iopub.execute_input":"2022-09-09T00:08:38.314070Z","iopub.status.idle":"2022-09-09T00:08:45.411416Z","shell.execute_reply.started":"2022-09-09T00:08:38.314040Z","shell.execute_reply":"2022-09-09T00:08:45.410342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All 3512 genes belonging to the same cell_id are adjacent.","metadata":{"papermill":{"duration":0.011125,"end_time":"2022-09-06T13:07:52.053324","exception":false,"start_time":"2022-09-06T13:07:52.042199","status":"completed"},"tags":[]}},{"cell_type":"code","source":"notallsame = 0\nfor id in range(0,3512):\n    ids = eval_ids.iloc[48663*140+id:65744180:3512]['cell_id']\n    if len(ids.unique()) >1:\n        notallsame += 1\nnotallsame","metadata":{"papermill":{"duration":40.447061,"end_time":"2022-09-06T13:08:32.511408","exception":false,"start_time":"2022-09-06T13:07:52.064347","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:08:45.413068Z","iopub.execute_input":"2022-09-09T00:08:45.413385Z","iopub.status.idle":"2022-09-09T00:09:04.212951Z","shell.execute_reply.started":"2022-09-09T00:08:45.413356Z","shell.execute_reply":"2022-09-09T00:09:04.211633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"But the gene_ids within 3512 are all different.","metadata":{"papermill":{"duration":0.010584,"end_time":"2022-09-06T13:08:32.532698","exception":false,"start_time":"2022-09-06T13:08:32.522114","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"In the comparison below, dummy_y is assumed to be infered multiome target, slected and sorted in accordance with evaluation_ids.csv","metadata":{"papermill":{"duration":0.010523,"end_time":"2022-09-06T13:08:32.554081","exception":false,"start_time":"2022-09-06T13:08:32.543558","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Method 1. Not convert eval_ids.gene_id to CategoricalDtype","metadata":{"papermill":{"duration":0.010603,"end_time":"2022-09-06T13:08:32.575557","exception":false,"start_time":"2022-09-06T13:08:32.564954","status":"completed"},"tags":[]}},{"cell_type":"code","source":"multi_start = 48663*140\neval_cells = eval_ids.iloc[multi_start::3512][[\"cell_id\"]]","metadata":{"papermill":{"duration":0.025418,"end_time":"2022-09-06T13:08:32.611816","exception":false,"start_time":"2022-09-06T13:08:32.586398","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:09:04.214209Z","iopub.execute_input":"2022-09-09T00:09:04.214576Z","iopub.status.idle":"2022-09-09T00:09:04.223135Z","shell.execute_reply.started":"2022-09-09T00:09:04.214543Z","shell.execute_reply":"2022-09-09T00:09:04.222153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_columns = pd.read_hdf(FP_MULTIOME_TRAIN_TARGETS, start=0, stop=10).columns","metadata":{"papermill":{"duration":0.14162,"end_time":"2022-09-06T13:08:32.764408","exception":false,"start_time":"2022-09-06T13:08:32.622788","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:09:04.226166Z","iopub.execute_input":"2022-09-09T00:09:04.226975Z","iopub.status.idle":"2022-09-09T00:09:04.336395Z","shell.execute_reply.started":"2022-09-09T00:09:04.226929Z","shell.execute_reply":"2022-09-09T00:09:04.335166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(y_columns)\ndisplay(eval_ids.iloc[:,1])","metadata":{"papermill":{"duration":0.028661,"end_time":"2022-09-06T13:08:32.804178","exception":false,"start_time":"2022-09-06T13:08:32.775517","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:09:04.340159Z","iopub.execute_input":"2022-09-09T00:09:04.340834Z","iopub.status.idle":"2022-09-09T00:09:04.353155Z","shell.execute_reply.started":"2022-09-09T00:09:04.340798Z","shell.execute_reply":"2022-09-09T00:09:04.352297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnp.random.seed(seed=0)\ndummy_y = pd.DataFrame( np.random.rand(eval_cells.shape[0],len(y_columns)),\n                      columns = y_columns)\ntgt = np.zeros((eval_ids.shape[0]))\nmulti_start = 48663*140\nlidx = 0\nfor idx in range(multi_start,eval_ids.shape[0],3512):\n    tgt_gene = eval_ids.iloc[idx:idx+3512,1]\n    tgt[idx:idx+3512] = dummy_y[dummy_y.index==lidx][tgt_gene]\n    lidx+=1\ndummy_y[dummy_y.index==lidx-1][tgt_gene]","metadata":{"papermill":{"duration":74.670983,"end_time":"2022-09-06T13:09:47.486548","exception":false,"start_time":"2022-09-06T13:08:32.815565","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:09:04.354731Z","iopub.execute_input":"2022-09-09T00:09:04.355121Z","iopub.status.idle":"2022-09-09T00:09:59.723385Z","shell.execute_reply.started":"2022-09-09T00:09:04.355088Z","shell.execute_reply":"2022-09-09T00:09:59.722303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"papermill":{"duration":0.257138,"end_time":"2022-09-06T13:09:47.755859","exception":false,"start_time":"2022-09-06T13:09:47.498721","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:09:59.724967Z","iopub.execute_input":"2022-09-09T00:09:59.725436Z","iopub.status.idle":"2022-09-09T00:09:59.876623Z","shell.execute_reply.started":"2022-09-09T00:09:59.725368Z","shell.execute_reply":"2022-09-09T00:09:59.875491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Method 2. Convert eval_ids.gene_id to CategoricalDtype","metadata":{"papermill":{"duration":0.012318,"end_time":"2022-09-06T13:09:47.780185","exception":false,"start_time":"2022-09-06T13:09:47.767867","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\neval_ids.gene_id = eval_ids.gene_id.astype(pd.CategoricalDtype())\n    ","metadata":{"papermill":{"duration":17.759281,"end_time":"2022-09-06T13:10:05.551448","exception":false,"start_time":"2022-09-06T13:09:47.792167","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:09:59.877981Z","iopub.execute_input":"2022-09-09T00:09:59.878360Z","iopub.status.idle":"2022-09-09T00:10:12.017880Z","shell.execute_reply.started":"2022-09-09T00:09:59.878330Z","shell.execute_reply":"2022-09-09T00:10:12.016798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(y_columns)\ndisplay(eval_ids.iloc[:,1])","metadata":{"papermill":{"duration":0.045107,"end_time":"2022-09-06T13:10:05.608706","exception":false,"start_time":"2022-09-06T13:10:05.563599","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:10:12.019450Z","iopub.execute_input":"2022-09-09T00:10:12.019779Z","iopub.status.idle":"2022-09-09T00:10:12.047993Z","shell.execute_reply.started":"2022-09-09T00:10:12.019749Z","shell.execute_reply":"2022-09-09T00:10:12.046671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnp.random.seed(seed=0)\ndummy_y = pd.DataFrame( np.random.rand(eval_cells.shape[0],len(y_columns)),\n#                      columns = cati_gene)  \n                      columns = y_columns)  \ntgt = np.zeros((eval_ids.shape[0]))\nmulti_start = 48663*140\nlidx = 0\nfor idx in range(multi_start,eval_ids.shape[0],3512):\n    tgt_gene = eval_ids.iloc[idx:idx+3512,1]\n    tgt[idx:idx+3512] = dummy_y[dummy_y.index==lidx][tgt_gene]\n    lidx+=1\ndummy_y[dummy_y.index==lidx-1][tgt_gene]","metadata":{"papermill":{"duration":77.744028,"end_time":"2022-09-06T13:11:23.365023","exception":false,"start_time":"2022-09-06T13:10:05.620995","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:10:12.049624Z","iopub.execute_input":"2022-09-09T00:10:12.050065Z","iopub.status.idle":"2022-09-09T00:11:07.438114Z","shell.execute_reply.started":"2022-09-09T00:10:12.050024Z","shell.execute_reply":"2022-09-09T00:11:07.437080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"papermill":{"duration":0.177796,"end_time":"2022-09-06T13:11:23.554974","exception":false,"start_time":"2022-09-06T13:11:23.377178","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:11:07.439601Z","iopub.execute_input":"2022-09-09T00:11:07.439894Z","iopub.status.idle":"2022-09-09T00:11:07.572627Z","shell.execute_reply.started":"2022-09-09T00:11:07.439866Z","shell.execute_reply":"2022-09-09T00:11:07.571264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Method 3. Convert eval_ids.gene_id to CategoricalDtype and columns to CategoricalIndex, and use reindex","metadata":{"papermill":{"duration":0.012528,"end_time":"2022-09-06T13:11:23.580451","exception":false,"start_time":"2022-09-06T13:11:23.567923","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\neval_ids.gene_id = eval_ids.gene_id.astype(pd.CategoricalDtype())\n","metadata":{"papermill":{"duration":0.185091,"end_time":"2022-09-06T13:11:23.778058","exception":false,"start_time":"2022-09-06T13:11:23.592967","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:27:17.397873Z","iopub.execute_input":"2022-09-09T00:27:17.398304Z","iopub.status.idle":"2022-09-09T00:27:31.185671Z","shell.execute_reply.started":"2022-09-09T00:27:17.398270Z","shell.execute_reply":"2022-09-09T00:27:31.184537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_columns = pd.read_hdf(FP_MULTIOME_TRAIN_TARGETS, start=0, stop=10).columns\ny_columns = pd.CategoricalIndex(y_columns, dtype=eval_ids.gene_id.dtype, name='gene_id')","metadata":{"papermill":{"duration":0.142195,"end_time":"2022-09-06T13:11:23.965170","exception":false,"start_time":"2022-09-06T13:11:23.822975","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:11:07.741763Z","iopub.execute_input":"2022-09-09T00:11:07.742197Z","iopub.status.idle":"2022-09-09T00:11:07.851842Z","shell.execute_reply.started":"2022-09-09T00:11:07.742156Z","shell.execute_reply":"2022-09-09T00:11:07.850835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(y_columns)\ndisplay(eval_ids.iloc[:,1])","metadata":{"papermill":{"duration":0.051321,"end_time":"2022-09-06T13:11:24.029095","exception":false,"start_time":"2022-09-06T13:11:23.977774","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:11:07.852963Z","iopub.execute_input":"2022-09-09T00:11:07.853271Z","iopub.status.idle":"2022-09-09T00:11:07.878429Z","shell.execute_reply.started":"2022-09-09T00:11:07.853245Z","shell.execute_reply":"2022-09-09T00:11:07.877661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nnp.random.seed(seed=0)\ndummy_y = pd.DataFrame( np.random.rand(eval_cells.shape[0],len(y_columns)),\n                      columns = y_columns)\ntgt = np.zeros((eval_ids.shape[0]))\nmulti_start = 48663*140\nidx = multi_start\nfor i,(index,row) in enumerate(dummy_y.iterrows()):\n    tgt_gene = eval_ids.iloc[idx:idx+3512,1]\n    tgt[idx:idx+3512] = row.reindex(tgt_gene)\n    idx += 3512\ndummy_y[dummy_y.index==lidx-1][tgt_gene]","metadata":{"papermill":{"duration":592.387097,"end_time":"2022-09-06T13:21:16.436593","exception":false,"start_time":"2022-09-06T13:11:24.049496","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-09-09T00:11:07.879695Z","iopub.execute_input":"2022-09-09T00:11:07.880207Z","iopub.status.idle":"2022-09-09T00:20:10.148148Z","shell.execute_reply.started":"2022-09-09T00:11:07.880175Z","shell.execute_reply":"2022-09-09T00:20:10.146996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creation of Submission file","metadata":{}},{"cell_type":"code","source":"%%time\nsubmission = pd.DataFrame(tgt,columns = ['target'])\nsubmission.index.name='row_id'\nsubmission = submission.round(6)\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-09-09T00:28:27.726583Z","iopub.execute_input":"2022-09-09T00:28:27.727349Z","iopub.status.idle":"2022-09-09T00:28:28.853593Z","shell.execute_reply.started":"2022-09-09T00:28:27.727309Z","shell.execute_reply":"2022-09-09T00:28:28.852457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.013636,"end_time":"2022-09-06T13:21:16.463632","exception":false,"start_time":"2022-09-06T13:21:16.449996","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.012391,"end_time":"2022-09-06T13:21:16.489246","exception":false,"start_time":"2022-09-06T13:21:16.476855","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}