{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-08T17:44:26.816465Z","iopub.execute_input":"2024-03-08T17:44:26.816850Z","iopub.status.idle":"2024-03-08T17:44:28.072545Z","shell.execute_reply.started":"2024-03-08T17:44:26.816820Z","shell.execute_reply":"2024-03-08T17:44:28.071292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install rdkit category_encoders captum portia-grn","metadata":{"execution":{"iopub.status.busy":"2024-03-08T13:14:48.395144Z","iopub.execute_input":"2024-03-08T13:14:48.395645Z","iopub.status.idle":"2024-03-08T13:14:48.401196Z","shell.execute_reply.started":"2024-03-08T13:14:48.395609Z","shell.execute_reply":"2024-03-08T13:14:48.400139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nimport math\nimport random\nimport warnings\nimport collections\nfrom typing import Dict, List, Tuple, Optional, Generator\nimport matplotlib.pyplot as plt\nimport matplotlib\nimport numpy as np\nimport scipy.stats\nimport scipy.sparse\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder, OneHotEncoder\nfrom sklearn.decomposition import KernelPCA\nfrom sklearn.metrics import matthews_corrcoef, roc_auc_score, average_precision_score\n","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:44:33.121655Z","iopub.execute_input":"2024-03-08T17:44:33.122372Z","iopub.status.idle":"2024-03-08T17:44:33.979560Z","shell.execute_reply.started":"2024-03-08T17:44:33.122315Z","shell.execute_reply":"2024-03-08T17:44:33.977972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data location\nDATA_FOLDER = '/kaggle/input/open-problems-single-cell-perturbations/'\n\n# Cell types for which all compounds are available in training set\nTRAIN_CELL_TYPES = ['NK cells', 'T cells CD4+', 'T cells CD8+', 'T regulatory cells']\n\n# Cell types for which some compounds are available in test set\nTEST_CELL_TYPES = ['Myeloid cells', 'B cells']\n\n# All cell types\nCELL_TYPES = TRAIN_CELL_TYPES + TEST_CELL_TYPES\n\n# Load DE data, as well as the cell type, compound and SMILES for each row\ndf = pd.read_parquet(os.path.join(DATA_FOLDER, 'de_train.parquet'))\ncell_types = df['cell_type']\nsm_names = df['sm_name']\nall_smiles = df['SMILES']\n\n# Convert Pandas dataframe to NumPy array\nfor col_name in ['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'control']:\n    df.drop(col_name, axis=1, inplace=True)\ndata = df.to_numpy(dtype=float)\n\n# Gene names, ordered by columns\ngene_names = list(df.columns)\ngene_name_dict = {gene_name: i for i, gene_name in enumerate(gene_names)}\n\n# Unique compound names and cell types\nunique_sm_names = list(set(sm_names))\nunique_cell_types = list(set(cell_types))\n\n# Dict to map compounds to their SMILES\nsm_name_to_smiles = {sm_name: smiles for sm_name, smiles in zip(sm_names, all_smiles)}\n\n# Make a dict version of the data matrix, to easily access the DE values\n# given the cell type and compound\ndata_dict = {}\nfor p in range(len(sm_names)):\n    if sm_names[p] not in data_dict:\n        data_dict[sm_names[p]] = {}\n    data_dict[sm_names[p]][cell_types[p]] = data[p, :]\n\n# Identify compounds for which all cell types are available\ntrain_sm_names = set()\nfor sm_name in data_dict.keys():\n    if len(data_dict[sm_name]) == len(unique_cell_types):\n        train_sm_names.add(sm_name)\n\n# Count, for each (cell_type, compound) pair, how many cells were used to infer the DE values\ndf = pd.read_csv(os.path.join(DATA_FOLDER, 'adata_obs_meta.csv'), header='infer', delimiter=',')\ncell_count_dict = collections.Counter(zip(df['cell_type'], df['sm_name']))\ncell_count_per_donor_dict = collections.Counter(zip(df['cell_type'], df['sm_name'], df['donor_id']))\ncell_counts = np.asarray([cell_count_dict[key] for key in zip(cell_types, sm_names)])","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:46:17.493666Z","iopub.execute_input":"2024-03-08T17:46:17.494148Z","iopub.status.idle":"2024-03-08T17:46:21.540743Z","shell.execute_reply.started":"2024-03-08T17:46:17.494112Z","shell.execute_reply":"2024-03-08T17:46:21.539606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_comparison(cell_type_a: str, cell_type_b: str, sm_names_: List[str]) -> None:\n    \"\"\"Show the correlations between 2 cell types for each pair of \"training\" compound,\n    using a heatmap.\n    \n    Args:\n        cell_type_a: First cell type (x-axis).\n        cell_type_b: Second cell type (y-axis).\n        sm_names_: List of compounds to be compared, pairwise.\n    \"\"\"\n    corr = np.ones((len(train_sm_names), len(train_sm_names)))\n    for i in range(corr.shape[0]):\n        for j in range(corr.shape[1]):\n            # Compute Pearson correlation coefficient\n            corr[i, j] = scipy.stats.pearsonr(\n                data_dict[sm_names_[i]][cell_type_a],\n                data_dict[sm_names_[j]][cell_type_b],\n            )[0]\n    plt.imshow(corr, cmap='PuOr', vmin=-1, vmax=1)\n    plt.xlabel(cell_type_a)\n    plt.ylabel(cell_type_b)\n    plt.xticks(range(len(sm_names_)), sm_names_, rotation=90)\n    plt.yticks(range(len(sm_names_)), sm_names_)\n    plt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:46:34.570062Z","iopub.execute_input":"2024-03-08T17:46:34.570538Z","iopub.status.idle":"2024-03-08T17:46:34.581921Z","shell.execute_reply.started":"2024-03-08T17:46:34.570495Z","shell.execute_reply":"2024-03-08T17:46:34.580663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_de_train = pd.read_parquet(os.path.join(DATA_FOLDER, 'de_train.parquet'))\nplt.figure(figsize = (20,4) )\nv = df_de_train.iloc[:,5:].max(axis = 0 ).sort_values(ascending = False, key = abs )\nplt.plot(v.values,'*-')\nplt.title('Max-abs DE for genes',fontsize = 20 )\nplt.grid()\nplt.show()\n# display(v.head(15))\nt = v\nn =0;d=15;\nfor n in [0,d,2*d,3*d,4*d,5*d,6*d,7*d,8*d,9*d]:\n    t2 = t.iloc[n:n+d].to_frame().reset_index()\n    t2.columns = ['Gene','DE']\n    if n == 0:\n        t3 = t2.copy()\n    else:\n        t3 = pd.concat( (t3,t2),axis = 1 )\n        \ndisplay(t3.round(1))","metadata":{"execution":{"iopub.status.busy":"2024-03-09T05:29:02.782458Z","iopub.execute_input":"2024-03-09T05:29:02.782922Z","iopub.status.idle":"2024-03-09T05:29:04.738382Z","shell.execute_reply.started":"2024-03-09T05:29:02.782873Z","shell.execute_reply":"2024-03-09T05:29:04.736061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(' \"Most often\" (compound/cell-types averaged) affected genes ')\nt =  (-(df_de_train.iloc[:,5:].T.abs())).rank().median(axis=1) # Use ranks and average (over pairs - compound/cell-types) them . Minus - to get top gene as rank=1 not as rank=len   \nt = t.sort_values()\nprint(' \"Most often\" (compound/cell-types averaged) affected genes ')\nn =0;d=15;\nfor n in [0,d,2*d,3*d,4*d,5*d,6*d,7*d,8*d,9*d]:\n    t2 = t.iloc[n:n+d].to_frame().reset_index()\n    t2.columns = ['Gene','Rank']\n    if n == 0:\n        t3 = t2.copy()\n    else:\n        t3 = pd.concat( (t3,t2),axis = 1 )\ndisplay(t3)\n\nplt.figure(figsize = (20,4) )\nplt.plot(t.values,'*')\nplt.ylabel('Rank (averaged) ',fontsize = 15)\nplt.grid()\nplt.title( '  \"Most often\" (compound/cell-types averaged) affected genes ', fontsize = 20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-09T05:29:25.117577Z","iopub.execute_input":"2024-03-09T05:29:25.118184Z","iopub.status.idle":"2024-03-09T05:29:29.682032Z","shell.execute_reply.started":"2024-03-09T05:29:25.118138Z","shell.execute_reply":"2024-03-09T05:29:29.680686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Sneak peak of the cell count data used to infer the DE values\n##(Computed from the metadata of the provided dataset)\ncell_counts","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:46:37.490438Z","iopub.execute_input":"2024-03-08T17:46:37.490859Z","iopub.status.idle":"2024-03-08T17:46:37.503373Z","shell.execute_reply.started":"2024-03-08T17:46:37.490828Z","shell.execute_reply":"2024-03-08T17:46:37.502141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Alignment With Biological Literature","metadata":{}},{"cell_type":"markdown","source":"[Peripheral Blood Mononuclear Cells](#ref_verhoeckx) have the following cell type frequencies, assuming myeloid cells include mostly monocytes and myeloid dendritic cells, and assuming regulatory T cells account for 5-10% of CD4+ T cells:\n- CD4+ T cells: 33-51%\n- CD8+ T cells: 16-26%\n- NK cells: 3-18%\n- B cells: 3-9%\n- Myeloid cells: 11-22%\n- Regulatory T cells: 1-5%","metadata":{}},{"cell_type":"code","source":"total_counts_per_donor = {'donor_0': {}, 'donor_1': {}, 'donor_2': {}}\nfor donor_id in total_counts_per_donor.keys():\n    for cell_type in CELL_TYPES:\n        total_counts_per_donor[donor_id][cell_type] = sum([cell_count_per_donor_dict[(cell_type, sm_name, donor_id)] for sm_name in train_sm_names])\n    \nBOUNDS = {'NK cells': (0.03, 0.18), 'T cells CD4+': (0.33, 0.51), 'T cells CD8+': (0.16, 0.26), 'T regulatory cells': (0.01, 0.05), 'Myeloid cells': (0.11, 0.22), 'B cells': (0.03, 0.09)}\nplt.figure(figsize=(14, 4))\nax = plt.subplot(1, 1, 1)\nwidth = 0.15\nplt.bar(np.arange(len(CELL_TYPES))-0.3, [BOUNDS[cell_type][0] for cell_type in CELL_TYPES], label='Minimum', width=width, color='mediumseagreen')\nfor i in range(3):\n    color = ['goldenrod', 'darkslateblue', 'rosybrown'][i]\n    proportions = [total_counts_per_donor[f'donor_{i}'][cell_type] / sum(total_counts_per_donor[f'donor_{i}'].values()) for cell_type in CELL_TYPES]\n    plt.bar(np.arange(len(CELL_TYPES))-0.15+0.15*i, proportions, label=f'donor_{i}', width=width, color=color)\nplt.bar(np.arange(len(CELL_TYPES))+0.3, [BOUNDS[cell_type][1] for cell_type in CELL_TYPES], label='Maximum', width=width, color='darkcyan')\nplt.xlabel('Cell types')\nplt.xticks(np.arange(len(CELL_TYPES)), CELL_TYPES)\nplt.ylabel('Total proportion')\nax.spines[['right', 'top']].set_visible(False)\nplt.grid(alpha=0.4, linestyle='--', linewidth=0.5, color='grey')\nprop = {'family': 'Century gothic', 'size': 9}\nplt.legend(prop=prop, bbox_to_anchor=(1.02, 1), loc='upper left', borderaxespad=0, frameon=False)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:47:14.997288Z","iopub.execute_input":"2024-03-08T17:47:14.997708Z","iopub.status.idle":"2024-03-08T17:47:15.417857Z","shell.execute_reply.started":"2024-03-08T17:47:14.997678Z","shell.execute_reply":"2024-03-08T17:47:15.416689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****These proportions are mostly consistent with the literature, \nexceptions being a very high proportion of B cells in donor 0, a high proportion of natural killer cells in donor 3, and low proportions of CD8+ T cells in all donors.****","metadata":{}},{"cell_type":"markdown","source":"## Correlation Plots","metadata":{}},{"cell_type":"code","source":"# data_dict","metadata":{"execution":{"iopub.status.busy":"2024-03-08T12:52:36.680110Z","iopub.execute_input":"2024-03-08T12:52:36.680546Z","iopub.status.idle":"2024-03-08T12:52:36.710375Z","shell.execute_reply.started":"2024-03-08T12:52:36.680500Z","shell.execute_reply":"2024-03-08T12:52:36.709256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Checking similarity between small molecules on the same cell type","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 22.5))\nfor i, cell_type in enumerate(CELL_TYPES):\n    plt.subplot(3, 2, i + 1)\n    plot_comparison(cell_type, cell_type, list(train_sm_names))\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:47:32.684997Z","iopub.execute_input":"2024-03-08T17:47:32.685453Z","iopub.status.idle":"2024-03-08T17:47:42.038974Z","shell.execute_reply.started":"2024-03-08T17:47:32.685416Z","shell.execute_reply":"2024-03-08T17:47:42.038063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It appears that some compounds induce very similar cell responses, and this can best be seen in T regulatory cells. In particular, Belinostat, Dabrafenib, Foretinib, MLN 2238, Crizotinib and Idelalisib all highly correlate between each other. Most of them are used in cancer treatment, and are meant for inhibiting protein kinases.","metadata":{}},{"cell_type":"markdown","source":"### Checking similarity between two cell types to the same perturbation","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15, 15))\nfor i, cell_type in enumerate(TRAIN_CELL_TYPES):\n    plt.subplot(2, 2, i + 1)\n    plot_comparison('B cells', cell_type, list(train_sm_names))\nplt.tight_layout()\nplt.show()\n\n\nplt.figure(figsize=(15, 15))\nfor i, cell_type in enumerate(TRAIN_CELL_TYPES):\n    plt.subplot(2, 2, i + 1)\n    plot_comparison('Myeloid cells', cell_type, list(train_sm_names))\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:48:28.496315Z","iopub.execute_input":"2024-03-08T17:48:28.497511Z","iopub.status.idle":"2024-03-08T17:48:40.218724Z","shell.execute_reply.started":"2024-03-08T17:48:28.497462Z","shell.execute_reply":"2024-03-08T17:48:40.217358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cell Counts vs the Postive differentially expressed values of the genes","metadata":{}},{"cell_type":"markdown","source":"The method Limma uses linear models to adjust for confounding factors like plate and donor. Since these factors are categorical, they need to be encoded into binary variables, resulting in a large number of covariates. When the number of samples is lower than the number of covariates, it can lead to unreliable results.\n\nTo address this, authors investigated if the number of cells affects the differential expression (DE) results, especially the fraction of genes showing positive DE. Normally, this fraction should remain constant regardless of the number of cells used.","metadata":{}},{"cell_type":"markdown","source":"### Categorisation based on cell type\n","metadata":{}},{"cell_type":"code","source":"CELL_TYPES","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:49:46.838783Z","iopub.execute_input":"2024-03-08T17:49:46.839277Z","iopub.status.idle":"2024-03-08T17:49:46.846954Z","shell.execute_reply.started":"2024-03-08T17:49:46.839228Z","shell.execute_reply":"2024-03-08T17:49:46.845659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask = (cell_types == 'NK cells')","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:49:49.854414Z","iopub.execute_input":"2024-03-08T17:49:49.854855Z","iopub.status.idle":"2024-03-08T17:49:49.861411Z","shell.execute_reply.started":"2024-03-08T17:49:49.854821Z","shell.execute_reply":"2024-03-08T17:49:49.860069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_counts[mask]","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:49:52.446954Z","iopub.execute_input":"2024-03-08T17:49:52.448362Z","iopub.status.idle":"2024-03-08T17:49:52.457195Z","shell.execute_reply.started":"2024-03-08T17:49:52.448307Z","shell.execute_reply":"2024-03-08T17:49:52.455785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.mean(data > 0, axis=1)[mask]","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:49:55.879229Z","iopub.execute_input":"2024-03-08T17:49:55.879668Z","iopub.status.idle":"2024-03-08T17:49:55.908978Z","shell.execute_reply.started":"2024-03-08T17:49:55.879634Z","shell.execute_reply":"2024-03-08T17:49:55.907730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:49:58.638177Z","iopub.execute_input":"2024-03-08T17:49:58.638648Z","iopub.status.idle":"2024-03-08T17:49:58.649459Z","shell.execute_reply.started":"2024-03-08T17:49:58.638611Z","shell.execute_reply":"2024-03-08T17:49:58.648351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nax = plt.subplot(1, 1, 1)\nfor cell_type in CELL_TYPES:\n    mask = (cell_types == cell_type)  # One separate color for each cell type\n    plt.scatter(cell_counts[mask], np.mean(data > 0, axis=1)[mask], label=cell_type, alpha=0.4)\nplt.xlabel('Cell count')\nplt.ylabel('Proportion of genes with DE > 0')\nplt.xscale('log')\nax.spines[['right', 'top']].set_visible(False)\nplt.grid(alpha=0.4, linestyle='--', linewidth=0.5, color='grey')\nprop = {'family': 'Century gothic', 'size': 9}\nplt.legend(prop=prop, bbox_to_anchor=(1.02, 1), loc='upper left', borderaxespad=0, frameon=False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:50:02.127923Z","iopub.execute_input":"2024-03-08T17:50:02.128353Z","iopub.status.idle":"2024-03-08T17:50:03.336646Z","shell.execute_reply.started":"2024-03-08T17:50:02.128320Z","shell.execute_reply":"2024-03-08T17:50:03.335561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- **There's a consistent negative linear relationship between cell counts and the proportion of genes showing positive differential expression for each cell type.**\n- **Each cell type seems to split into two clusters, with stronger separation seen when cell counts are lower. This suggests that these clusters may not have a biological basis.**\n","metadata":{}},{"cell_type":"markdown","source":"### Categorisation based on compound","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nxs = cell_counts\nys = np.mean(data > 0, axis=1)  # Proportion of genes with strictly positive DE\nax = plt.subplot(1, 1, 1)\n\nfor sm_name in train_sm_names:\n    # One separate color for compound\n    mask = np.asarray([sm_name == x for x in sm_names], dtype=bool)\n    plt.scatter(xs[mask], ys[mask], alpha=0.4, label=sm_name)\n\nplt.xlabel('Cell count')\nplt.ylabel('Proportion of genes with DE > 0')\nplt.xscale('log')\nax.spines[['right', 'top']].set_visible(False)\nplt.grid(alpha=0.4, linestyle='--', linewidth=0.5, color='grey')\nprop = {'family': 'Century gothic', 'size': 9}\nplt.legend(prop=prop, bbox_to_anchor=(1.02, 1), loc='upper left', borderaxespad=0, frameon=False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:51:15.719133Z","iopub.execute_input":"2024-03-08T17:51:15.719644Z","iopub.status.idle":"2024-03-08T17:51:16.758457Z","shell.execute_reply.started":"2024-03-08T17:51:15.719608Z","shell.execute_reply":"2024-03-08T17:51:16.757169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Standard deviation of DE as a function of cell count for each cell type","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nxs = cell_counts\nys = np.std(np.abs(data), axis=1)\nax = plt.subplot(1, 1, 1)\nfor cell_type in CELL_TYPES:\n    mask = (cell_types == cell_type)\n    plt.scatter(xs[mask], ys[mask], label=cell_type, alpha=0.4)\nplt.xlabel('Cell count')\nplt.ylabel('DE standard deviation')\nplt.xscale('log')\nax.spines[['right', 'top']].set_visible(False)\nplt.grid(alpha=0.4, linestyle='--', linewidth=0.5, color='grey')\nprop = {'family': 'Century gothic', 'size': 9}\nplt.legend(prop=prop, bbox_to_anchor=(1.02, 1), loc='upper left', borderaxespad=0, frameon=False)\n#plt.axvline(x=25, linestyle='--', linewidth=1, color='black')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-08T17:52:04.132982Z","iopub.execute_input":"2024-03-08T17:52:04.133404Z","iopub.status.idle":"2024-03-08T17:52:04.995823Z","shell.execute_reply.started":"2024-03-08T17:52:04.133373Z","shell.execute_reply":"2024-03-08T17:52:04.994922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Strong volatility in standard deviation of DE values across genes observed for pseudo-bulked samples with extreme cell counts(low or high)\n- High cell counts (positive controls which our case is) likely reflect real biological signals.\n- Unexpected variability in low cell counts due to:\n  - Limited experimental evidence supporting significant results.\n- Typically lower volatility observed for cell counts in the range of 25-1000 (Fairly consistent).\n","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}