{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-16T19:42:41.255973Z","iopub.execute_input":"2023-09-16T19:42:41.256392Z","iopub.status.idle":"2023-09-16T19:42:41.623745Z","shell.execute_reply.started":"2023-09-16T19:42:41.256361Z","shell.execute_reply":"2023-09-16T19:42:41.622822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## This notebook will explore some initial dataset exploration to undertand cell type differences within cellular samples treated with some popular small molecules\nOne of the objectives of this competition is to determine how small molecules are involved in changes in gene expression within cells that can possibly impact cell type differences within samples. I will first explore the dataset and how some popular drugs have an impact on the cell type distribution in the samples.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\ndf = pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/adata_train.parquet')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:42:41.625605Z","iopub.execute_input":"2023-09-16T19:42:41.626165Z","iopub.status.idle":"2023-09-16T19:43:58.731386Z","shell.execute_reply.started":"2023-09-16T19:42:41.626129Z","shell.execute_reply":"2023-09-16T19:43:58.730079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n!pip install scanpy\nimport scanpy as sc\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:43:58.733207Z","iopub.execute_input":"2023-09-16T19:43:58.733668Z","iopub.status.idle":"2023-09-16T19:44:24.657469Z","shell.execute_reply.started":"2023-09-16T19:43:58.733625Z","shell.execute_reply":"2023-09-16T19:44:24.656113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf2 = pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet')\ndf2.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:44:24.661489Z","iopub.execute_input":"2023-09-16T19:44:24.662219Z","iopub.status.idle":"2023-09-16T19:44:27.520445Z","shell.execute_reply.started":"2023-09-16T19:44:24.662183Z","shell.execute_reply":"2023-09-16T19:44:27.519070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df20 = df2.drop(df2.columns[[2,3,4,5]], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:44:27.522199Z","iopub.execute_input":"2023-09-16T19:44:27.522587Z","iopub.status.idle":"2023-09-16T19:44:27.574544Z","shell.execute_reply.started":"2023-09-16T19:44:27.522554Z","shell.execute_reply":"2023-09-16T19:44:27.573084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df20","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:44:27.576678Z","iopub.execute_input":"2023-09-16T19:44:27.577123Z","iopub.status.idle":"2023-09-16T19:44:27.626002Z","shell.execute_reply.started":"2023-09-16T19:44:27.577089Z","shell.execute_reply":"2023-09-16T19:44:27.624713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df7 = df2.cell_type.value_counts()\ndf7.plot(kind='pie', title = \"Cell Type Distribution in  All Samples\")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:44:27.627840Z","iopub.execute_input":"2023-09-16T19:44:27.628241Z","iopub.status.idle":"2023-09-16T19:44:27.893781Z","shell.execute_reply.started":"2023-09-16T19:44:27.628205Z","shell.execute_reply":"2023-09-16T19:44:27.892494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df8 = df2.sm_name.value_counts()\ndf8.describe","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:44:27.895522Z","iopub.execute_input":"2023-09-16T19:44:27.896333Z","iopub.status.idle":"2023-09-16T19:44:27.909914Z","shell.execute_reply.started":"2023-09-16T19:44:27.896285Z","shell.execute_reply":"2023-09-16T19:44:27.908295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2.describe()","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:44:27.912199Z","iopub.execute_input":"2023-09-16T19:44:27.913077Z","iopub.status.idle":"2023-09-16T19:45:09.963162Z","shell.execute_reply.started":"2023-09-16T19:44:27.913028Z","shell.execute_reply":"2023-09-16T19:45:09.961844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df4 = df2[[\"cell_type\", \"sm_name\"]]\ndf4","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:45:09.968965Z","iopub.execute_input":"2023-09-16T19:45:09.969438Z","iopub.status.idle":"2023-09-16T19:45:09.986004Z","shell.execute_reply.started":"2023-09-16T19:45:09.969401Z","shell.execute_reply":"2023-09-16T19:45:09.984457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The Scatter plot below shows the distribution of cell types accordring to small molecule name\nAlthough the y-axis is crowded becuase of 146 small molecules being tested in this dataset, we can obsereve that NK, CD4+ T cells, CD8+ T cells and Tregs are in all samples of the small molecule treated samples being tested. There are a subset of small molecule samples treated that also have B and myeloid cells.","metadata":{}},{"cell_type":"code","source":"\ndf4.plot.scatter('cell_type', 'sm_name'); # The ';' is to avoid showing a message before showing the plot","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:45:09.987743Z","iopub.execute_input":"2023-09-16T19:45:09.988162Z","iopub.status.idle":"2023-09-16T19:45:11.396469Z","shell.execute_reply.started":"2023-09-16T19:45:09.988129Z","shell.execute_reply":"2023-09-16T19:45:11.395416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## [I would like to acknowledge Alexander Chervov's notebook and will explore how the following top drugs can impact sample type distribution](https://www.kaggle.com/code/alexandervc/op2-eda-baseline-s)\n\nhttps://www.kaggle.com/code/alexandervc/op2-eda-baseline-s\n\nlist_top_drugs = ['MLN 2238', 'Resminostat', 'CEP-18770 (Delanzomib)', 'Oprozomib (ONX 0912)', 'Belinostat', 'Vorinostat', 'Ganetespib (STA-9090)', 'Scriptaid', 'Proscillaridin A;Proscillaridin-A', 'Alvocidib', 'IN1451']\n","metadata":{}},{"cell_type":"code","source":"df21 = df4[df4['sm_name'].str.contains('Riociguat')]\ndf21\ndf18 = df21.cell_type.value_counts()\ndf18.plot(kind='pie', title = \"Cell Type Distribution in Samples treated with Riocigat \")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:45:11.397948Z","iopub.execute_input":"2023-09-16T19:45:11.398335Z","iopub.status.idle":"2023-09-16T19:45:11.586411Z","shell.execute_reply.started":"2023-09-16T19:45:11.398302Z","shell.execute_reply":"2023-09-16T19:45:11.584719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df21 = df4[df4['sm_name'].str.contains('MLN 2238')]\ndf21\ndf18 = df21.cell_type.value_counts()\ndf18.plot(kind='pie', title = \"Cell Type Distribution in Samples treated with MLN2238 \")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:45:11.589101Z","iopub.execute_input":"2023-09-16T19:45:11.590210Z","iopub.status.idle":"2023-09-16T19:45:11.832283Z","shell.execute_reply.started":"2023-09-16T19:45:11.590148Z","shell.execute_reply":"2023-09-16T19:45:11.830550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df21 = df4[df4['sm_name'].str.contains('Belinostat')]\ndf21\ndf18 = df21.cell_type.value_counts()\ndf18.plot(kind='pie', title = \"Cell Type Distribution in Samples treated with Belinostat\")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:45:11.835350Z","iopub.execute_input":"2023-09-16T19:45:11.837307Z","iopub.status.idle":"2023-09-16T19:45:12.062645Z","shell.execute_reply.started":"2023-09-16T19:45:11.837237Z","shell.execute_reply":"2023-09-16T19:45:12.060889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"df21 = df4[df4['sm_name'].str.contains('Vorinostat')]\ndf21\ndf18 = df21.cell_type.value_counts()\ndf18.plot(kind='pie', title = \"Cell Type Distribution in Samples treated with Vorinostat \")","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:45:12.065151Z","iopub.execute_input":"2023-09-16T19:45:12.066821Z","iopub.status.idle":"2023-09-16T19:45:12.254093Z","shell.execute_reply.started":"2023-09-16T19:45:12.066751Z","shell.execute_reply":"2023-09-16T19:45:12.252312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Next Step working on GeneExpression Exploration in different cell types","metadata":{}},{"cell_type":"code","source":"df23 = df20[df20['cell_type'].str.contains('NK cells')]\ndf23","metadata":{"execution":{"iopub.status.busy":"2023-09-16T19:52:30.418996Z","iopub.execute_input":"2023-09-16T19:52:30.419556Z","iopub.status.idle":"2023-09-16T19:52:30.488459Z","shell.execute_reply.started":"2023-09-16T19:52:30.419518Z","shell.execute_reply":"2023-09-16T19:52:30.487147Z"},"trusted":true},"execution_count":null,"outputs":[]}]}