{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Open Problems in Cell Analyis: Quick EDA","metadata":{}},{"cell_type":"markdown","source":"<div style=\"color:white;\n       display:fill;\n       border-radius:5px;\n       background-color:#ffffe6;\n       font-size:120%;\">\n    <p style=\"padding: 10px;\n          color:black;\">\nLet's take a look at CITEseq inputs first\n    </p>\n</div>","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.set()\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-16T16:40:40.917455Z","iopub.execute_input":"2022-08-16T16:40:40.918020Z","iopub.status.idle":"2022-08-16T16:40:42.029563Z","shell.execute_reply.started":"2022-08-16T16:40:40.917838Z","shell.execute_reply":"2022-08-16T16:40:42.028229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --quiet tables","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:40:42.031686Z","iopub.execute_input":"2022-08-16T16:40:42.032051Z","iopub.status.idle":"2022-08-16T16:40:55.747188Z","shell.execute_reply.started":"2022-08-16T16:40:42.032018Z","shell.execute_reply":"2022-08-16T16:40:55.745744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.makedirs('/kaggle/working/inputs', exist_ok=True)\n# Circumvent read-only issues\n!cp ../input/open-problems-multimodal/train_cite_inputs.h5 '/kaggle/working/inputs'","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:40:55.748979Z","iopub.execute_input":"2022-08-16T16:40:55.749380Z","iopub.status.idle":"2022-08-16T16:41:22.311628Z","shell.execute_reply.started":"2022-08-16T16:40:55.749342Z","shell.execute_reply":"2022-08-16T16:41:22.308615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with pd.HDFStore('/kaggle/working/inputs/train_cite_inputs.h5') as data:\n    shape = data['/train_cite_inputs'].shape\n    print(f\"There are {shape[0]} cell IDs and {shape[1]} columns (!)\")\n    selected_columns = data['/train_cite_inputs'].columns[:40]\n    # We select only 50 cells for starters\n    df = data['/train_cite_inputs'][selected_columns].head(50)\n    \ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:41:22.317318Z","iopub.execute_input":"2022-08-16T16:41:22.317836Z","iopub.status.idle":"2022-08-16T16:44:08.302105Z","shell.execute_reply.started":"2022-08-16T16:41:22.317789Z","shell.execute_reply":"2022-08-16T16:44:08.301113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --quiet joypy","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:44:08.303873Z","iopub.execute_input":"2022-08-16T16:44:08.304532Z","iopub.status.idle":"2022-08-16T16:44:20.978773Z","shell.execute_reply.started":"2022-08-16T16:44:08.304497Z","shell.execute_reply":"2022-08-16T16:44:20.977420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport joypy\nimport numpy as np\n\ndef color_gradient(x=0.0, start=(0, 0, 0), stop=(1, 1, 1)):\n    r = np.interp(x, [0, 1], [start[0], stop[0]])\n    g = np.interp(x, [0, 1], [start[1], stop[1]])\n    b = np.interp(x, [0, 1], [start[2], stop[2]])\n    return (r, g, b)\n\njoypy.joyplot(\n              df,\n    title=\"Cell distribution by gene\",overlap=4,\n              colormap=lambda x: color_gradient(x, start=(153/256, 255/256, 204/256),\n                                                stop=(204/256, 102/256, 255/256)),\n              linecolor='black', linewidth=.5,\n             figsize=(7,12),);\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:44:20.981246Z","iopub.execute_input":"2022-08-16T16:44:20.981705Z","iopub.status.idle":"2022-08-16T16:44:23.835501Z","shell.execute_reply.started":"2022-08-16T16:44:20.981663Z","shell.execute_reply":"2022-08-16T16:44:23.834632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr = df.corr()\nplt.figure(figsize=(12,8));\nsns.heatmap(corr, cmap=\"viridis\");","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:44:23.836928Z","iopub.execute_input":"2022-08-16T16:44:23.837524Z","iopub.status.idle":"2022-08-16T16:44:24.959378Z","shell.execute_reply.started":"2022-08-16T16:44:23.837488Z","shell.execute_reply":"2022-08-16T16:44:24.958079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"----\n<div style=\"color:white;\n       display:fill;\n       border-radius:5px;\n       background-color:#ffffe6;\n       font-size:120%;\">\n    <p style=\"padding: 10px;\n          color:black;\">\nNow we get to the CITEseq targets\n    </p>\n</div>","metadata":{}},{"cell_type":"code","source":"os.makedirs('/kaggle/working/labels', exist_ok=True)\n!cp ../input/open-problems-multimodal/train_cite_targets.h5 '/kaggle/working/labels'","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:44:24.961245Z","iopub.execute_input":"2022-08-16T16:44:24.962019Z","iopub.status.idle":"2022-08-16T16:44:27.065858Z","shell.execute_reply.started":"2022-08-16T16:44:24.961970Z","shell.execute_reply":"2022-08-16T16:44:27.064425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with pd.HDFStore('/kaggle/working/labels/train_cite_targets.h5') as data:\n    shape = data['/train_cite_targets'].shape\n    print(f\"There are {shape[0]} cell IDs and {shape[1]} columns (!)\")\n    selected_columns = data['/train_cite_targets'].columns[:40]\n    # We select only 50 cells for starters\n    df_targets = data['/train_cite_targets'][selected_columns].head(50)","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:44:27.068559Z","iopub.execute_input":"2022-08-16T16:44:27.068966Z","iopub.status.idle":"2022-08-16T16:44:27.729126Z","shell.execute_reply.started":"2022-08-16T16:44:27.068931Z","shell.execute_reply":"2022-08-16T16:44:27.728034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_targets.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:44:27.732457Z","iopub.execute_input":"2022-08-16T16:44:27.733116Z","iopub.status.idle":"2022-08-16T16:44:27.768154Z","shell.execute_reply.started":"2022-08-16T16:44:27.733079Z","shell.execute_reply":"2022-08-16T16:44:27.766980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joypy.joyplot(\n              df_targets,\n    title=\"Cell distribution by surface protein\",overlap=4,\n              colormap=lambda x: color_gradient(x, start=(153/256, 255/256, 204/256),\n                                                stop=(204/256, 102/256, 255/256)),\n              linecolor='black', linewidth=.5,\n             figsize=(7,12),);","metadata":{"execution":{"iopub.status.busy":"2022-08-16T16:44:27.769489Z","iopub.execute_input":"2022-08-16T16:44:27.769911Z","iopub.status.idle":"2022-08-16T16:44:30.592082Z","shell.execute_reply.started":"2022-08-16T16:44:27.769878Z","shell.execute_reply":"2022-08-16T16:44:30.590864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}