{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport gc\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n!pip install tables","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-26T12:32:31.432102Z","iopub.execute_input":"2022-08-26T12:32:31.432591Z","iopub.status.idle":"2022-08-26T12:32:48.231063Z","shell.execute_reply.started":"2022-08-26T12:32:31.432498Z","shell.execute_reply":"2022-08-26T12:32:48.229802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_df = pd.read_csv(\"../input/open-problems-multimodal/metadata.csv\")\nprint(\"Number of cellids:\", len(meta_df))\nmeta_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-26T12:32:48.233687Z","iopub.execute_input":"2022-08-26T12:32:48.234069Z","iopub.status.idle":"2022-08-26T12:32:48.627066Z","shell.execute_reply.started":"2022-08-26T12:32:48.234026Z","shell.execute_reply":"2022-08-26T12:32:48.625814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"Count of Samples seen as factor of technolgoy\")\nsns.countplot(data=meta_df, x='technology')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-26T12:32:48.628957Z","iopub.execute_input":"2022-08-26T12:32:48.629585Z","iopub.status.idle":"2022-08-26T12:32:49.030850Z","shell.execute_reply.started":"2022-08-26T12:32:48.629540Z","shell.execute_reply":"2022-08-26T12:32:49.029493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"Distribution of samples collected by Day\")\nsns.countplot(data=meta_df, x='day', hue='technology')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-26T12:32:49.034665Z","iopub.execute_input":"2022-08-26T12:32:49.035101Z","iopub.status.idle":"2022-08-26T12:32:49.397485Z","shell.execute_reply.started":"2022-08-26T12:32:49.035065Z","shell.execute_reply":"2022-08-26T12:32:49.396170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=meta_df, x='donor', hue='technology')","metadata":{"execution":{"iopub.status.busy":"2022-08-26T12:32:49.398781Z","iopub.execute_input":"2022-08-26T12:32:49.399546Z","iopub.status.idle":"2022-08-26T12:32:49.733164Z","shell.execute_reply.started":"2022-08-26T12:32:49.399510Z","shell.execute_reply":"2022-08-26T12:32:49.731895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. ","metadata":{}},{"cell_type":"markdown","source":"# checking the sparsity of the gene_id in inputs and outputs of Multiome Dataset","metadata":{}},{"cell_type":"code","source":"def get_sparse_percent_from_dataframe(df):\n    v = (df==0).sum(axis=1)\n    cellids = list(v.index)\n    sparse_percent = v.values/df.shape[1]\n    \n    return pd.DataFrame.from_dict({\n        'cell_id': cellids,\n        'sparse_percent' : sparse_percent\n    })","metadata":{"execution":{"iopub.status.busy":"2022-08-26T12:33:27.291576Z","iopub.execute_input":"2022-08-26T12:33:27.293753Z","iopub.status.idle":"2022-08-26T12:33:27.301477Z","shell.execute_reply.started":"2022-08-26T12:33:27.293687Z","shell.execute_reply":"2022-08-26T12:33:27.300285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_sparse_percent_data(filepath, technology):\n    sparse_data = []\n    batch_size = 1000\n\n    for i in range(0, len(meta_df[meta_df.technology==technology]), 1000):\n        train_multi_inputs = pd.read_hdf(filepath, start=i, stop=i+1000)\n        sparse_data.append( get_sparse_percent_from_dataframe(train_multi_inputs) )\n        del train_multi_inputs\n        gc.collect()\n\n    sparse_data = pd.concat(sparse_data)\n    return sparse_data","metadata":{"execution":{"iopub.status.busy":"2022-08-26T12:33:27.303760Z","iopub.execute_input":"2022-08-26T12:33:27.304102Z","iopub.status.idle":"2022-08-26T12:33:27.316786Z","shell.execute_reply.started":"2022-08-26T12:33:27.304065Z","shell.execute_reply":"2022-08-26T12:33:27.315477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"number of inputs in Multiome Data:\", pd.read_hdf(\"../input/open-problems-multimodal/train_multi_inputs.h5\", start=0, stop=1).shape[1])\nprint(\"number of target in Multiome Data:\", pd.read_hdf(\"../input/open-problems-multimodal/train_multi_targets.h5\", start=0, stop=1).shape[1])\n\n\nprint(\"number of inputs in CITE Data:\", pd.read_hdf(\"../input/open-problems-multimodal/train_cite_inputs.h5\", start=0, stop=1).shape[1])\nprint(\"number of target in CITE Data:\", pd.read_hdf(\"../input/open-problems-multimodal/train_cite_targets.h5\", start=0, stop=1).shape[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-26T12:48:08.502818Z","iopub.execute_input":"2022-08-26T12:48:08.503275Z","iopub.status.idle":"2022-08-26T12:48:09.503058Z","shell.execute_reply.started":"2022-08-26T12:48:08.503236Z","shell.execute_reply":"2022-08-26T12:48:09.501628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sparse_multiome_input  = get_sparse_percent_data(\"../input/open-problems-multimodal/train_multi_inputs.h5\", 'multiome')\nsparse_multiome_target = get_sparse_percent_data(\"../input/open-problems-multimodal/train_multi_targets.h5\", 'multiome')\n","metadata":{"execution":{"iopub.status.busy":"2022-08-26T12:33:27.318931Z","iopub.execute_input":"2022-08-26T12:33:27.320708Z","iopub.status.idle":"2022-08-26T12:35:29.885736Z","shell.execute_reply.started":"2022-08-26T12:33:27.320659Z","shell.execute_reply":"2022-08-26T12:35:29.884533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sparse_cite_input  = get_sparse_percent_data(\"../input/open-problems-multimodal/train_cite_inputs.h5\", 'multiome')\nsparse_cite_target = get_sparse_percent_data(\"../input/open-problems-multimodal/train_cite_targets.h5\", 'multiome')","metadata":{"execution":{"iopub.status.busy":"2022-08-26T12:41:44.569892Z","iopub.execute_input":"2022-08-26T12:41:44.570841Z","iopub.status.idle":"2022-08-26T12:42:13.076045Z","shell.execute_reply.started":"2022-08-26T12:41:44.570794Z","shell.execute_reply":"2022-08-26T12:42:13.074861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax=plt.subplots(2, 2, figsize=(13, 7), )\n\n\nax[0][0].hist(sparse_multiome_input.sparse_percent, bins=100)\nax[0, 1].hist(sparse_multiome_target.sparse_percent, bins=100)\n\nax[1, 0].hist(sparse_cite_input.sparse_percent, bins=100)\nax[1, 1].hist(sparse_cite_target.sparse_percent, bins=100)\n\nax[0, 0].set_title(\"Amount of sparsity in the inputs of Multiome data\")\nax[0, 1].set_title(\"Amount of sparsity in the targets of Multiome data\")\n\n\nax[1, 0].set_title(\"Amount of sparsity in the inputs of CITE data\")\nax[1, 1].set_title(\"Amount of sparsity in the targets of CITE data\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-26T12:44:12.016463Z","iopub.execute_input":"2022-08-26T12:44:12.016889Z","iopub.status.idle":"2022-08-26T12:44:13.109146Z","shell.execute_reply.started":"2022-08-26T12:44:12.016854Z","shell.execute_reply":"2022-08-26T12:44:13.107799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}