{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"code","source":"!pip install hdf5plugin~=2.0","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:05:11.594190Z","iopub.execute_input":"2022-09-26T09:05:11.594664Z","iopub.status.idle":"2022-09-26T09:05:25.487232Z","shell.execute_reply.started":"2022-09-26T09:05:11.594626Z","shell.execute_reply":"2022-09-26T09:05:25.485494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport h5py\nimport hdf5plugin\nimport gc\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom csv import writer\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-09-26T11:28:15.441440Z","iopub.execute_input":"2022-09-26T11:28:15.443756Z","iopub.status.idle":"2022-09-26T11:28:15.461327Z","shell.execute_reply.started":"2022-09-26T11:28:15.443615Z","shell.execute_reply":"2022-09-26T11:28:15.457800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nSUBMISSON = os.path.join(DATA_DIR,\"sample_submission.csv\")\nEVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:05:26.832916Z","iopub.execute_input":"2022-09-26T09:05:26.833828Z","iopub.status.idle":"2022-09-26T09:05:26.842181Z","shell.execute_reply.started":"2022-09-26T09:05:26.833790Z","shell.execute_reply":"2022-09-26T09:05:26.840941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Goal of This Notebook\nThis notebook is based on [@shuntarotanaka](https://www.kaggle.com/code/shuntarotanaka)'s notebook.\nThe major difference is as follows.\n* I explicitly split the submission data by the training dataset type (Multiome vs Citeseq) instead of supplementing the predicted values from two models (one for multione and the other for citeseq) to it. \n* I decided to get the average of each `gene_id` by `cell_type`.","metadata":{}},{"cell_type":"markdown","source":"# Splitting Submission Index by Training Dataset\nAccording to the \"Data\" description, it says that\n> To facilitate submission scoring, we only require predictions on a subset of the Multiome data. This subset was created by sampling 30% of the Multiome rows, and for each row, 15% of the columns. The sample of columns varies from row-to-row. All of the CITEseq labels are scored.\n\nHonestly, I could not understand what is written literally until I looked through some submitted codes by those fore-running Kagglers and check evaluation_ids. <br>\n\nIn a nutshell, `submission.csv` can be splitted to two: 1) predicted values based on Multiome training dataset and 2) predicted values based on Citeseq training dataset.<br>\nIt is possible that chromatin accessibility affects the protein level to some degree, so you may argue that those two cannot be split. However, seeing many Kagglers building a model from Multiome and another from Citeseq and combining them, I decided to explicitly split submission data by which training dataset the row refers to.<br>\n\n","metadata":{}},{"cell_type":"code","source":"eval_df = pd.read_csv(EVALUATION_IDS, index_col='row_id')\neval_df.index.name = None\ndisplay(eval_df)\neval_df_length = eval_df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:05:26.845369Z","iopub.execute_input":"2022-09-26T09:05:26.846828Z","iopub.status.idle":"2022-09-26T09:07:28.629074Z","shell.execute_reply.started":"2022-09-26T09:05:26.846789Z","shell.execute_reply":"2022-09-26T09:07:28.627638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I noticed first 5 `gene_id` start from \"CD\" and the last gene ids start from \"ENSG\". \nStudying a bit about Gene Naming Convension at [genenames.org](https://www.genenames.org/data/gene-symbol-report/#!/hgnc_id/HGNC:1338), I learned that those starting from \"ENSG\" are coming from [Ensembl](https://en.wikipedia.org/wiki/Ensembl_genome_database_project) those starting from \"CD\" are classified as Gene Symbols.\n","metadata":{}},{"cell_type":"markdown","source":"Next, I extract the gene ids used for `train_cite_targets.h5` and `train_multi_targets.h5`.","metadata":{}},{"cell_type":"code","source":"def convert_h5_to_numpy(f_path):\n    f = h5py.File(f_path, 'r')\n    f_name = f_path.split('/')[-1].split('.')[0]\n    \n    group = f[f_name]\n    \n    X_columns = np.char.decode(group['axis0'][:])\n    X_index = np.char.decode(group['axis1'][:])\n    X_data = group['block0_values']\n    \n    return X_columns, X_index, X_data","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:28.631183Z","iopub.execute_input":"2022-09-26T09:07:28.632057Z","iopub.status.idle":"2022-09-26T09:07:28.640505Z","shell.execute_reply.started":"2022-09-26T09:07:28.632006Z","shell.execute_reply":"2022-09-26T09:07:28.639233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_targets_columns, train_cite_targets_index, train_cite_targets_data = convert_h5_to_numpy(FP_CITE_TRAIN_TARGETS)\n# print gene ids present in train_cite_targets\nprint(train_cite_targets_columns)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:28.642169Z","iopub.execute_input":"2022-09-26T09:07:28.643477Z","iopub.status.idle":"2022-09-26T09:07:28.715697Z","shell.execute_reply.started":"2022-09-26T09:07:28.643426Z","shell.execute_reply":"2022-09-26T09:07:28.714329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of them look like gene symbols but there are some exceptions such as 'Mouse-IgG2a'.","metadata":{}},{"cell_type":"code","source":"train_multi_targets_columns, train_multi_targets_index, train_multi_targets_data = convert_h5_to_numpy(FP_MULTIOME_TRAIN_TARGETS)\n# print gene ids present in train_multi_targets\nprint(train_multi_targets_columns)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:28.717470Z","iopub.execute_input":"2022-09-26T09:07:28.718199Z","iopub.status.idle":"2022-09-26T09:07:28.814016Z","shell.execute_reply.started":"2022-09-26T09:07:28.718154Z","shell.execute_reply":"2022-09-26T09:07:28.813034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The displayed ones are all look like an ensembl gene code.","metadata":{}},{"cell_type":"markdown","source":"Next, I am going to split `eval_df` based on the available `gene_id` in each dataset.","metadata":{}},{"cell_type":"code","source":"eval_df_cite = eval_df[eval_df['gene_id'].isin(train_cite_targets_columns)].copy()\nprint('evaluation_ids for Citeseq:')\ndisplay(eval_df_cite)\n\neval_df_multi = eval_df[eval_df['gene_id'].isin(train_multi_targets_columns)].copy()\nprint('evaluation_ids for Multiome:')\ndisplay(eval_df_multi)\n\n# Delete the original df to save memory\ndel eval_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:28.815394Z","iopub.execute_input":"2022-09-26T09:07:28.815762Z","iopub.status.idle":"2022-09-26T09:07:47.068947Z","shell.execute_reply.started":"2022-09-26T09:07:28.815729Z","shell.execute_reply":"2022-09-26T09:07:47.068134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Just to make sure no records are missing, check if the total record count is the same before and after the data split.","metadata":{}},{"cell_type":"code","source":"assert eval_df_length == eval_df_cite.shape[0] + eval_df_multi.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:47.069971Z","iopub.execute_input":"2022-09-26T09:07:47.070298Z","iopub.status.idle":"2022-09-26T09:07:47.076065Z","shell.execute_reply.started":"2022-09-26T09:07:47.070266Z","shell.execute_reply":"2022-09-26T09:07:47.075053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loading Metadata","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:31:40.539251Z","iopub.execute_input":"2022-08-11T21:31:40.540195Z","iopub.status.idle":"2022-08-11T21:31:40.544877Z","shell.execute_reply.started":"2022-08-11T21:31:40.540149Z","shell.execute_reply":"2022-08-11T21:31:40.543637Z"}}},{"cell_type":"code","source":"cell_meta_df = pd.read_csv(FP_CELL_METADATA, index_col='cell_id')\n# Remove index name\ncell_meta_df.index.name = None\n# Display Top 5\ndisplay(cell_meta_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:47.080278Z","iopub.execute_input":"2022-09-26T09:07:47.080634Z","iopub.status.idle":"2022-09-26T09:07:47.507094Z","shell.execute_reply.started":"2022-09-26T09:07:47.080599Z","shell.execute_reply":"2022-09-26T09:07:47.505971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check how many records exist by the dataset type\ncell_meta_df.groupby(['technology']).size()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:47.508296Z","iopub.execute_input":"2022-09-26T09:07:47.509126Z","iopub.status.idle":"2022-09-26T09:07:47.540996Z","shell.execute_reply.started":"2022-09-26T09:07:47.509091Z","shell.execute_reply":"2022-09-26T09:07:47.540129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As I did to `eval_df` in the previous step, I want to completely separate the data of citeseq and multiome. So I split the metadata too.","metadata":{}},{"cell_type":"code","source":"cell_meta_df_cite = cell_meta_df[cell_meta_df['technology'] == 'citeseq'].copy()\ncell_meta_df_multi = cell_meta_df[cell_meta_df['technology'] == 'multiome'].copy()\n\ndel cell_meta_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:47.542308Z","iopub.execute_input":"2022-09-26T09:07:47.542636Z","iopub.status.idle":"2022-09-26T09:07:47.723765Z","shell.execute_reply.started":"2022-09-26T09:07:47.542607Z","shell.execute_reply":"2022-09-26T09:07:47.722577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Computing Average by Cell Type: Citeseq","metadata":{}},{"cell_type":"markdown","source":"Before getting the average value, I am going to bring `cell_type` into `train_cite_targets_data` next.","metadata":{}},{"cell_type":"markdown","source":"To begin with, let's store `train_cite_targets_data` into a dataframe.","metadata":{}},{"cell_type":"code","source":"train_cite_targets_df = pd.DataFrame(data=train_cite_targets_data\n             , columns=train_cite_targets_columns\n             , index=train_cite_targets_index\n)\n# Display top 5\ndisplay(train_cite_targets_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:47.725426Z","iopub.execute_input":"2022-09-26T09:07:47.725778Z","iopub.status.idle":"2022-09-26T09:07:53.217900Z","shell.execute_reply.started":"2022-09-26T09:07:47.725748Z","shell.execute_reply":"2022-09-26T09:07:53.216724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then, bring `cell_type` from cell_meta_df_cite and join into `train_cite_targets_df`","metadata":{}},{"cell_type":"code","source":"train_cite_targets_df = train_cite_targets_df.join(cell_meta_df_cite[[\"cell_type\"]])\ntrain_cite_targets_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:53.219046Z","iopub.execute_input":"2022-09-26T09:07:53.219347Z","iopub.status.idle":"2022-09-26T09:07:53.309123Z","shell.execute_reply.started":"2022-09-26T09:07:53.219318Z","shell.execute_reply":"2022-09-26T09:07:53.308254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As shown above, I could pull `cell_type` from the metadata. So, I am going to take the average by `cell_type`.","metadata":{}},{"cell_type":"code","source":"train_cite_targets_df_mean = train_cite_targets_df.groupby([\"cell_type\"]).mean()\n# Remove index name\ntrain_cite_targets_df_mean.index.name = None\n# Display Top 5\ntrain_cite_targets_df_mean.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:53.310418Z","iopub.execute_input":"2022-09-26T09:07:53.310937Z","iopub.status.idle":"2022-09-26T09:07:53.420476Z","shell.execute_reply.started":"2022-09-26T09:07:53.310898Z","shell.execute_reply":"2022-09-26T09:07:53.419138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I decided to encode `cell_type` and `gene_id` to a numeric value for the better performance overall.","metadata":{}},{"cell_type":"code","source":"ordinal_encoder_cite_cell_type = OrdinalEncoder()\nordinal_encoder_cite_gene_id = OrdinalEncoder()\n\n# Convert the data shape to fit & transform the encoder\ncite_cell_types_for_encoder = train_cite_targets_df_mean.index.values.reshape(-1,1)\ncite_gene_ids_for_encoder = train_cite_targets_df_mean.columns.values.reshape(-1,1)\n# print(cite_cell_types_for_encoder)\n# print(cite_gene_ids_for_encoder)\n\n# Flatten the data and convert it to numeric\ntrain_cite_targets_df_mean.index = ordinal_encoder_cite_cell_type.fit_transform(cite_cell_types_for_encoder).flatten()\ntrain_cite_targets_df_mean.columns = ordinal_encoder_cite_gene_id.fit_transform(cite_gene_ids_for_encoder).flatten()\n\ndisplay(train_cite_targets_df_mean.head())","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:53.422018Z","iopub.execute_input":"2022-09-26T09:07:53.422479Z","iopub.status.idle":"2022-09-26T09:07:53.458010Z","shell.execute_reply.started":"2022-09-26T09:07:53.422445Z","shell.execute_reply":"2022-09-26T09:07:53.456646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Computing Average by Cell Type: Multiome\nThere is a challenge that the Multiome data is so huge that you cannot load it to the dataframe directly.","metadata":{}},{"cell_type":"code","source":"# Running this will crash the notebook\n# train_multi_targets_df = pd.DataFrame(data=train_multi_targets_data\n#              , columns=train_multi_targets_columns\n#              , index=train_multi_targets_index\n# )\n# # Display top 5\n# display(train_multi_targets_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:53.459502Z","iopub.execute_input":"2022-09-26T09:07:53.459845Z","iopub.status.idle":"2022-09-26T09:07:53.468446Z","shell.execute_reply.started":"2022-09-26T09:07:53.459814Z","shell.execute_reply":"2022-09-26T09:07:53.467432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Referring to how to [process huge numpy array in chunk](https://stackoverflow.com/questions/53972000/built-in-method-for-chunking-a-numpy-array), I define a method to create a generator.\nWith this, I attempt to take the average for every dataset of 5000 records.","metadata":{}},{"cell_type":"code","source":"def create_chunk_df_generator(target_index, target_column, target_data, chunksize):\n    \n    # Check if the index size matches with the data\n    assert target_index.shape[0] == target_data.shape[0]\n    \n    for start in range(0, len(target_data), chunksize):\n        chunk_data = target_data[start:start + chunksize]\n        chunk_index = target_index[start:start + chunksize]\n        \n        chunk_df = pd.DataFrame(data=chunk_data\n             , columns=target_column\n             , index=chunk_index\n        )\n        \n        yield chunk_df","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:53.470246Z","iopub.execute_input":"2022-09-26T09:07:53.470670Z","iopub.status.idle":"2022-09-26T09:07:53.482037Z","shell.execute_reply.started":"2022-09-26T09:07:53.470637Z","shell.execute_reply":"2022-09-26T09:07:53.480558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Create an empty df that will be filled with the average value\ntrain_multi_targets_df_mean = pd.DataFrame(columns=train_multi_targets_columns)\n\n# Define chunksize\nchunksize = 5000\n\n# For-in loop using the generator creation method\nfor train_multi_targets_chunk_df in create_chunk_df_generator(train_multi_targets_index, train_multi_targets_columns, train_multi_targets_data, chunksize):\n    train_multi_targets_chunk_df = train_multi_targets_chunk_df.join(cell_meta_df_multi[[\"cell_type\"]])\n    train_multi_targets_chunk_df_mean = train_multi_targets_chunk_df.groupby([\"cell_type\"]).mean()    \n    train_multi_targets_df_mean = pd.concat([train_multi_targets_df_mean, train_multi_targets_chunk_df_mean])\n    # Remove comment below if you want to check how the data looks like\n    # break","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:07:53.483874Z","iopub.execute_input":"2022-09-26T09:07:53.484293Z","iopub.status.idle":"2022-09-26T09:09:58.061560Z","shell.execute_reply.started":"2022-09-26T09:07:53.484259Z","shell.execute_reply":"2022-09-26T09:09:58.057782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The outcome `train_multi_targets_df_mean` contains the average of every 5000 records, so that I could reduce the data volume.\nNext, I take the average of the entire data by averaging the values further.","metadata":{}},{"cell_type":"code","source":"train_multi_targets_df_mean = train_multi_targets_df_mean.groupby(train_multi_targets_df_mean.index).mean()\n# Display the df\ndisplay(train_multi_targets_df_mean)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:09:58.063928Z","iopub.execute_input":"2022-09-26T09:09:58.064354Z","iopub.status.idle":"2022-09-26T09:09:58.368202Z","shell.execute_reply.started":"2022-09-26T09:09:58.064312Z","shell.execute_reply":"2022-09-26T09:09:58.367065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Submission Data for Multiome contains \"hidden\" cell_type. \nFor that, I am going to prepare a record for it as well.","metadata":{}},{"cell_type":"code","source":"multi_total_average_series = train_multi_targets_chunk_df_mean.mean().rename('hidden')\ntrain_multi_targets_df_mean = train_multi_targets_df_mean.append(multi_total_average_series)\ndisplay(train_multi_targets_df_mean)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:09:58.369887Z","iopub.execute_input":"2022-09-26T09:09:58.370282Z","iopub.status.idle":"2022-09-26T09:09:58.470960Z","shell.execute_reply.started":"2022-09-26T09:09:58.370249Z","shell.execute_reply":"2022-09-26T09:09:58.469799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Convert `cell_id` and `gene_id` of the Multiome data likewise.","metadata":{}},{"cell_type":"code","source":"ordinal_encoder_multi_cell_type = OrdinalEncoder()\nordinal_encoder_multi_gene_id = OrdinalEncoder()\n\n# Convert the data shape to fit & transform the encoder\nmulti_cell_types_for_encoder = train_multi_targets_df_mean.index.values.reshape(-1,1)\nmulti_gene_ids_for_encoder = train_multi_targets_df_mean.columns.values.reshape(-1,1)\n# print(multi_cell_types_for_encoder)\n# print(multi_gene_ids_for_encoder)\n\n# Flatten the data\ntrain_multi_targets_df_mean.index = ordinal_encoder_multi_cell_type.fit_transform(multi_cell_types_for_encoder).flatten()\ntrain_multi_targets_df_mean.columns = ordinal_encoder_multi_gene_id.fit_transform(multi_gene_ids_for_encoder).flatten()\n\ndisplay(train_multi_targets_df_mean)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:09:58.472213Z","iopub.execute_input":"2022-09-26T09:09:58.472554Z","iopub.status.idle":"2022-09-26T09:09:58.635041Z","shell.execute_reply.started":"2022-09-26T09:09:58.472522Z","shell.execute_reply":"2022-09-26T09:09:58.633893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating Submit File","metadata":{}},{"cell_type":"code","source":"submission_df = pd.read_csv(SUBMISSON, usecols=['row_id'], index_col='row_id')","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:09:58.636545Z","iopub.execute_input":"2022-09-26T09:09:58.636915Z","iopub.status.idle":"2022-09-26T09:11:02.351338Z","shell.execute_reply.started":"2022-09-26T09:09:58.636882Z","shell.execute_reply":"2022-09-26T09:11:02.350116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's separate the submission data according to the data source and add necessary columns to obtain the target value.","metadata":{}},{"cell_type":"code","source":"submission_df_cite = pd.merge(submission_df, eval_df_cite, left_index=True, right_index=True)\nsubmission_df_cite = pd.merge(submission_df_cite, cell_meta_df_cite[[\"cell_type\"]], left_on='cell_id', right_index=True)\ndisplay(submission_df_cite.head())\n\nsubmission_df_multi = pd.merge(submission_df, eval_df_multi, left_index=True, right_index=True)\nsubmission_df_multi = pd.merge(submission_df_multi, cell_meta_df_multi[[\"cell_type\"]], left_on='cell_id', right_index=True)\ndisplay(submission_df_multi.head())\n\ndel submission_df","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:11:02.352818Z","iopub.execute_input":"2022-09-26T09:11:02.353154Z","iopub.status.idle":"2022-09-26T09:11:38.794351Z","shell.execute_reply.started":"2022-09-26T09:11:02.353123Z","shell.execute_reply":"2022-09-26T09:11:38.793136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Then, convert `gene_id` and `cell_type` using OrdinalEncoder.","metadata":{}},{"cell_type":"code","source":"submission_df_cite.gene_id = ordinal_encoder_cite_gene_id.transform(submission_df_cite.gene_id.values.reshape(-1, 1)).flatten()\nsubmission_df_cite.cell_type = ordinal_encoder_cite_cell_type.transform(submission_df_cite.cell_type.values.reshape(-1, 1)).flatten()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:11:38.795743Z","iopub.execute_input":"2022-09-26T09:11:38.796097Z","iopub.status.idle":"2022-09-26T09:11:42.244446Z","shell.execute_reply.started":"2022-09-26T09:11:38.796064Z","shell.execute_reply":"2022-09-26T09:11:42.243247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nsubmission_df_cite.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:11:42.246079Z","iopub.execute_input":"2022-09-26T09:11:42.246414Z","iopub.status.idle":"2022-09-26T09:11:42.379859Z","shell.execute_reply.started":"2022-09-26T09:11:42.246383Z","shell.execute_reply":"2022-09-26T09:11:42.378674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm --d submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:11:42.381244Z","iopub.execute_input":"2022-09-26T09:11:42.381578Z","iopub.status.idle":"2022-09-26T09:11:43.622214Z","shell.execute_reply.started":"2022-09-26T09:11:42.381536Z","shell.execute_reply":"2022-09-26T09:11:43.620846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Create an empty submission csv that will be filled with the target value\npd.DataFrame(columns=['row_id', 'target']).to_csv('submission.csv', index=False, header=True)\n\nchunksize = 5000\nloading_size = 0\nloading_percent_progress = .00\nsubmission_df_cite_length = submission_df_cite.shape[0]\n\nwith tqdm(total=100.00) as pbar:\n    for chunk_df in create_chunk_df_generator(submission_df_cite.index.to_numpy(), submission_df_cite.columns.tolist(), submission_df_cite.to_numpy(), chunksize):\n        \n        target_list = []\n        for index, row in chunk_df.iterrows():\n            # Get the target value and append it to target_list\n            target_list.append(train_cite_targets_df_mean.at[row.cell_type, row.gene_id])\n        \n        chunk_df['target'] = np.array(target_list)\n\n        # append data frame to CSV file\n        chunk_df[['target']].to_csv('submission.csv', mode='a', index=True, header=False)\n\n        # Display loading bar using tqdm to make sure the cell is running \n        loading_size += chunksize\n        loading_percent_progress = (loading_size / submission_df_cite_length * 100) - loading_percent_progress\n        pbar.update(loading_percent_progress)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:11:43.628051Z","iopub.execute_input":"2022-09-26T09:11:43.628434Z","iopub.status.idle":"2022-09-26T09:21:02.721460Z","shell.execute_reply.started":"2022-09-26T09:11:43.628397Z","shell.execute_reply":"2022-09-26T09:21:02.720212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del submission_df_cite","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:21:02.723730Z","iopub.execute_input":"2022-09-26T09:21:02.724198Z","iopub.status.idle":"2022-09-26T09:21:02.731966Z","shell.execute_reply.started":"2022-09-26T09:21:02.724150Z","shell.execute_reply":"2022-09-26T09:21:02.730622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'm going to do the same for the Multiome data.","metadata":{}},{"cell_type":"code","source":"submission_df_multi.gene_id = ordinal_encoder_multi_gene_id.transform(submission_df_multi.gene_id.values.reshape(-1, 1)).flatten()\nsubmission_df_multi.cell_type = ordinal_encoder_multi_cell_type.transform(submission_df_multi.cell_type.values.reshape(-1, 1)).flatten()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:21:02.733581Z","iopub.execute_input":"2022-09-26T09:21:02.733959Z","iopub.status.idle":"2022-09-26T09:21:53.380238Z","shell.execute_reply.started":"2022-09-26T09:21:02.733918Z","shell.execute_reply":"2022-09-26T09:21:53.378796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()\nsubmission_df_multi.head()","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:21:53.382146Z","iopub.execute_input":"2022-09-26T09:21:53.382558Z","iopub.status.idle":"2022-09-26T09:21:53.580254Z","shell.execute_reply.started":"2022-09-26T09:21:53.382512Z","shell.execute_reply":"2022-09-26T09:21:53.578685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nchunksize = 5000\nloading_size = 0\nloading_percent_progress = .00\nsubmission_df_multi_length = submission_df_multi.shape[0]\n\nwith tqdm(total=100.00) as pbar:\n    for chunk_df in create_chunk_df_generator(submission_df_multi.index.to_numpy(), submission_df_multi.columns.tolist(), submission_df_multi.to_numpy(), chunksize):\n        \n        target_list = []\n        for index, row in chunk_df.iterrows():\n            # Get the target value and append it to target_list\n            target_list.append(train_multi_targets_df_mean.at[row.cell_type, row.gene_id])\n        \n        chunk_df['target'] = np.array(target_list)\n\n        # append data frame to CSV file\n        chunk_df[['target']].to_csv('submission.csv', mode='a', index=True, header=False)\n\n        # Display loading bar using tqdm to make sure the cell is running \n        loading_size += chunksize\n        loading_percent_progress = (loading_size / submission_df_multi_length * 100) - loading_percent_progress\n        pbar.update(loading_percent_progress)","metadata":{"execution":{"iopub.status.busy":"2022-09-26T09:21:53.582080Z","iopub.execute_input":"2022-09-26T09:21:53.582482Z","iopub.status.idle":"2022-09-26T10:43:43.551449Z","shell.execute_reply.started":"2022-09-26T09:21:53.582445Z","shell.execute_reply":"2022-09-26T10:43:43.549581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}