{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":7230206,"sourceType":"datasetVersion","datasetId":1608748}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about \n\nHere we look on housekeeping genes (list from: https://www.cell.com/trends/genetics/fulltext/S0168-9525(13)00089-9, Human housekeeping genes, Eli Eisenberg Erez Y. Levanon Published:June 28, 2013 )\n\nMotivated by findings:\nhttps://www.kaggle.com/competitions/open-problems-single-cell-perturbations/discussion/461159\n20th Place Solution Writeup For Open Problems - Single-cell\nPerturbations Competition\nAntoine Passemiers and Jalil Nourisa\n\n\nWho observed (and found ways to quantify) that housekeeping are less changed than generic genes - which is quite natural.\n\n\n\nWe also observe some sub-clusters in housekeeping genes \n","metadata":{}},{"cell_type":"markdown","source":"# Load list of housekeeping genes ( Eisenberg,  Levanon , 2013)\n\nHuman housekeeping genes, revisited\nEli Eisenberg\nErez Y. Levanon\nPublished:June 28, 2013\n\nKaggle dataset: 'Genes information' https://www.kaggle.com/datasets/alexandervc/genes-information/\ncollects various genes groups, lists, info, etc. ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n#fn = '/kaggle/input/genes-information/Human_housekeeping_genes_Eisenberg_Levanon_2013_TrendsGenetics.txt'\n# df_hk_genes = pd.read_csv(fn, header = None, sep = '\\t')\n# df_hk_genes[0] = [t.replace(' ','') for t in df_hk_genes[0]]\n# df_hk_genes.columns = ['Symbol', 'NM_id(RefSeq_mRNA_curated)']\nfn = '/kaggle/input/genes-information/Human_housekeeping_genes_Eisenberg_Levanon_2013_TrendsGenetics.csv'\ndf_hk_genes = pd.read_csv(fn,index_col = 0)\n# df_hk_genes.to_csv('df_hk_genes.csv')\ndf_hk_genes","metadata":{"execution":{"iopub.status.busy":"2023-12-18T13:01:24.607791Z","iopub.execute_input":"2023-12-18T13:01:24.608158Z","iopub.status.idle":"2023-12-18T13:01:24.643423Z","shell.execute_reply.started":"2023-12-18T13:01:24.608130Z","shell.execute_reply":"2023-12-18T13:01:24.642364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preliminaries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time() \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport pandas as pd\nimport tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-18T13:01:31.184146Z","iopub.execute_input":"2023-12-18T13:01:31.184523Z","iopub.status.idle":"2023-12-18T13:01:43.999633Z","shell.execute_reply.started":"2023-12-18T13:01:31.184495Z","shell.execute_reply":"2023-12-18T13:01:43.998523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_de_train = pd.read_parquet(fn)# , index_col = 0)\nprint(df_de_train.shape)\ndisplay(df_de_train )\n\nplt.figure(figsize = (20,4) )\nv = df_de_train.iloc[:,5:].max(axis = 0 ).sort_values(ascending = False, key = abs )\nplt.plot(v.values,'*-')\nplt.title('Max-abs DE for genes',fontsize = 20 )\nplt.grid()\nplt.show()\ndisplay(v.head(15))\n\n# %%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_id_map = pd.read_csv(fn,index_col = 0)\nprint(df_id_map.shape)\ndisplay(df_id_map)\nfn = '/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'\ndf_sample_submit = pd.read_csv(fn, index_col = 0)\nprint(df_sample_submit.shape)\ndisplay( df_sample_submit )\n\n\nfeatures_columns = ['cell_type','sm_name'] # \n# selected_features_columns = ['cell_type','sm_name']  # [ 'sm_name'] # What features to consider - others not used \n#     # For OP2 task we have two features - cell-type and compound - both categorical\n#     # Some simple models (like Ridge) better work when one uses only one feature - 'sm_name' (=\"compound\") - so you can try that option. \n#     # But seems pyboost is powerful enough to proper work with the both features.\n#     # Table with results for Ridge can be found here: https://www.kaggle.com/code/alexandervc/op2-target-encoders?scriptVersionId=148576642&cellId=1\n#     # Or here: https://docs.google.com/spreadsheets/d/1APN63PMaWZygVjYimK9Ivt0RvifdAU5JRYkxiDn4szw/edit?usp=sharing (sheet \"Encoders\") and discussed here:\n    \nX_submit_categorical = df_id_map[features_columns] # selected_features_columns] # pd.DataFrame(df_id_map, columns= features_columns )\nX_full_categorical = df_de_train[features_columns]# selected_features_columns ]\n\nY_full = df_de_train.iloc[:,5:].values\nprint('X_submit_categorical.shape, X_full_categorical.shape ,  Y_full.shape', X_submit_categorical.shape, X_full_categorical.shape ,  Y_full.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-18T13:01:44.001854Z","iopub.execute_input":"2023-12-18T13:01:44.003155Z","iopub.status.idle":"2023-12-18T13:01:50.189737Z","shell.execute_reply.started":"2023-12-18T13:01:44.003114Z","shell.execute_reply":"2023-12-18T13:01:50.188393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Look on intersection - 3518 out of 3804 - quite good ","metadata":{}},{"cell_type":"code","source":"l = df_hk_genes.iloc[:,0]\ns = set(l) & set(df_de_train.columns)\nlist_not_hk = list( set(df_de_train.columns[5:]) - set(l) )\nlist_hk = list(s)\nprint(list_hk[:10])\nlen(s), len(set(l) ), len(list_not_hk)\n","metadata":{"execution":{"iopub.status.busy":"2023-12-18T13:01:50.191281Z","iopub.execute_input":"2023-12-18T13:01:50.191712Z","iopub.status.idle":"2023-12-18T13:01:50.211259Z","shell.execute_reply.started":"2023-12-18T13:01:50.191672Z","shell.execute_reply":"2023-12-18T13:01:50.210162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Housekeeping genes are less affected than others\n\nAnother look on results of: \n\nhttps://www.kaggle.com/competitions/open-problems-single-cell-perturbations/discussion/461159\n20th Place Solution Writeup For Open Problems - Single-cell\nPerturbations Competition\nAntoine Passemiers and Jalil Nourisa\n","metadata":{}},{"cell_type":"code","source":"d = pd.DataFrame(); IX = -1\nlist_thresholds = [0.1,1e-2,1e-3,1e-4,1e-5,1e-6,1e-7,1e-8,1e-9,1e-10,1e-11,1e-12,1e-13,1e-14,1e-15,1e-16]\nfor t in list_thresholds:\n    v  = df_de_train[list_not_hk].values.ravel()\n    v = 10**(-np.abs(v))\n    m = v < t\n    #print(m.sum()/len(v) )\n    IX+=1\n    d.loc[IX, 'Threshold'] = t\n    d.loc[IX, 'Not housekeeping percent'] = 100*m.sum()/len(v)\n    \n    \n    v  = df_de_train[list_hk].values.ravel()\n    v = 10**(-np.abs(v))\n    m = v < t\n    #print(m.sum()/len(v) )\n    d.loc[IX, 'Housekeeping percent'] = 100*m.sum()/len(v)\nplt.figure(figsize = (15,4))    \nfor col in d.columns[1:]:\n    plt.plot(d[col].values , label = col )\nplt.grid()\nplt.legend()\nplt.xticks( )\nplt.xticks(range(len(list_thresholds)), list_thresholds)\nplt.xlabel('Threshold', fontsize = 20)\nplt.title('Percent DE expressed stronger than theshold on p-value', fontsize = 20)\nplt.show()\nd['Difference'] = d[d.columns[1]] - d[d.columns[2]] \nd","metadata":{"execution":{"iopub.status.busy":"2023-12-18T13:01:55.955916Z","iopub.execute_input":"2023-12-18T13:01:55.956722Z","iopub.status.idle":"2023-12-18T13:02:03.411812Z","shell.execute_reply.started":"2023-12-18T13:01:55.956687Z","shell.execute_reply":"2023-12-18T13:02:03.411038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# From direct visualization effect is not so clear seen","metadata":{}},{"cell_type":"code","source":"v  = df_de_train[list_not_hk].values.ravel()\nv = 10**(-np.abs(v))\nm = v < 0.01\nprint(m.sum()/len(v) )\nplt.hist(v, bins = 100)\nplt.show()\n\nv  = df_de_train[list_hk].values.ravel()\nv = 10**(-np.abs(v))\nm = v < 0.01\nprint(m.sum()/len(v) )\nplt.hist(v, bins = 100)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-18T13:02:08.388663Z","iopub.execute_input":"2023-12-18T13:02:08.389193Z","iopub.status.idle":"2023-12-18T13:02:09.772936Z","shell.execute_reply.started":"2023-12-18T13:02:08.389152Z","shell.execute_reply":"2023-12-18T13:02:09.771832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clusters in housekeeping genes","metadata":{}},{"cell_type":"code","source":"%%time\nd2 = df_de_train[list_hk]\ncm = d2.corr()\nsns.clustermap(cm)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-18T13:02:12.413741Z","iopub.execute_input":"2023-12-18T13:02:12.414137Z","iopub.status.idle":"2023-12-18T13:03:15.402207Z","shell.execute_reply.started":"2023-12-18T13:02:12.414108Z","shell.execute_reply":"2023-12-18T13:03:15.400977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Spearman","metadata":{}},{"cell_type":"code","source":"%%time\nd2 = df_de_train[list_hk]\ncm = d2.corr(method = 'spearman')\nsns.clustermap(cm)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-18T13:03:40.580844Z","iopub.execute_input":"2023-12-18T13:03:40.581748Z","iopub.status.idle":"2023-12-18T13:04:43.577058Z","shell.execute_reply.started":"2023-12-18T13:03:40.581714Z","shell.execute_reply":"2023-12-18T13:04:43.575930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Clustermap genes/samples  - no so good","metadata":{}},{"cell_type":"code","source":"%%time\nd2 = df_de_train[list_hk[:200]]\n# cm = d2.corr()\nsns.clustermap(d2)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:52:02.872445Z","iopub.execute_input":"2023-12-18T10:52:02.872868Z","iopub.status.idle":"2023-12-18T10:52:04.239936Z","shell.execute_reply.started":"2023-12-18T10:52:02.872833Z","shell.execute_reply":"2023-12-18T10:52:04.238836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Not about housekeeping genes , but some other eda\n","metadata":{}},{"cell_type":"markdown","source":"# Uniform distribution under null hypothesis\n\nBased on   https://www.kaggle.com/competitions/open-problems-single-cell-perturbations/discussion/461159\n\nJALIL NOURISA · 20TH IN THIS COMPETITION · POSTED 5 DAYS AGO\n","metadata":{}},{"cell_type":"code","source":"X = df_de_train.iloc[:,5:].values\nv = X.ravel()\nv = 10**(-np.abs(v))\nplt.hist(v,bins = 300)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:44:22.059525Z","iopub.execute_input":"2023-12-18T10:44:22.059974Z","iopub.status.idle":"2023-12-18T10:44:23.380455Z","shell.execute_reply.started":"2023-12-18T10:44:22.059930Z","shell.execute_reply":"2023-12-18T10:44:23.379553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for uv in df_de_train['cell_type'].unique():\n    m = df_de_train['cell_type'] == uv \n    X = df_de_train[m].iloc[:,5:].values\n    v = X.ravel()\n    v = 10**(-np.abs(v))\n    print(uv, m.sum(),  np.min(v))\n    plt.hist(v,bins = 100)\n    plt.title(uv, fontsize = 20 )\n    plt.show()\n    plt.hist(v[v<1e-5],bins = 100)\n    plt.title(uv, fontsize = 20 )\n    plt.show()    ","metadata":{"execution":{"iopub.status.busy":"2023-12-18T10:44:24.336211Z","iopub.execute_input":"2023-12-18T10:44:24.337010Z","iopub.status.idle":"2023-12-18T10:44:30.764207Z","shell.execute_reply.started":"2023-12-18T10:44:24.336947Z","shell.execute_reply":"2023-12-18T10:44:30.763038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_de_train['cell_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T21:03:56.660706Z","iopub.execute_input":"2023-12-17T21:03:56.661019Z","iopub.status.idle":"2023-12-17T21:03:56.669931Z","shell.execute_reply.started":"2023-12-17T21:03:56.66099Z","shell.execute_reply":"2023-12-17T21:03:56.668927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_de_train.iloc[:,5:]\nX.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.240654Z","iopub.execute_input":"2023-11-30T19:12:33.240913Z","iopub.status.idle":"2023-11-30T19:12:33.272957Z","shell.execute_reply.started":"2023-11-30T19:12:33.240893Z","shell.execute_reply":"2023-11-30T19:12:33.272095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['cell_type'] ==  'Myeloid cells'#  'B cells'\nv = X[m].values.ravel()\ns1 = pd.Series(v).describe()\ns1","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.274035Z","iopub.execute_input":"2023-11-30T19:12:33.27455Z","iopub.status.idle":"2023-11-30T19:12:33.301344Z","shell.execute_reply.started":"2023-11-30T19:12:33.274528Z","shell.execute_reply":"2023-11-30T19:12:33.300576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_df = []\nfor col in ['B cells' , 'Myeloid cells' ]:\n    m = df_de_train['cell_type'] ==  col\n    v = X[m].values.ravel()\n    s2 = pd.Series(v).describe()\n    s2.name = col\n    list_df.append( s2.to_frame() ) \nd = pd.concat( list_df, axis = 1)\nd","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.302172Z","iopub.execute_input":"2023-11-30T19:12:33.302399Z","iopub.status.idle":"2023-11-30T19:12:33.339917Z","shell.execute_reply.started":"2023-11-30T19:12:33.302381Z","shell.execute_reply":"2023-11-30T19:12:33.33918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in ['B cells' , 'Myeloid cells' ]:\n    m = df_de_train['cell_type'] ==  col\n    v = X[m].values.ravel()\n    plt.plot(v)\n    plt.title(col,fontsize = 20)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.340566Z","iopub.execute_input":"2023-11-30T19:12:33.340776Z","iopub.status.idle":"2023-11-30T19:12:33.82321Z","shell.execute_reply.started":"2023-11-30T19:12:33.340756Z","shell.execute_reply":"2023-11-30T19:12:33.822382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['cell_type'] == 'Myeloid cells' \ndf_de_train[m]['sm_name'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.82486Z","iopub.execute_input":"2023-11-30T19:12:33.825272Z","iopub.status.idle":"2023-11-30T19:12:33.834329Z","shell.execute_reply.started":"2023-11-30T19:12:33.825246Z","shell.execute_reply":"2023-11-30T19:12:33.833531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Got from that:\n# Drugs higher effect on myeloids:\nl6 = ['Crizotinib',\n'Dabrafenib',\n'R428',\n'Porcn Inhibitor III',\n'Foretinib',\n'Dactolisib']\n\n\nprint( col  )\n\nm = df_de_train['cell_type'] == 'Myeloid cells' \ni = 0;\nfor drug in df_de_train['sm_name'][m].unique(): \n    plt.figure(figsize = (20,3) )\n    plt.subplot(1,2,1)\n    i+=1\n    print(i, drug)\n    col = 'Myeloid cells'\n    m = df_de_train['cell_type'] == col\n    #plt.subplot(1,2,1)\n    m2 = df_de_train['sm_name'] == drug\n    m2 = m2 & m\n    plt.plot( X[m2].values.ravel() , label = col )\n    plt.title(drug + ' ' + col  ,fontsize = 20)\n    \n    #plt.subplot(1,2,2)\n    col = 'B cells'\n    m = df_de_train['cell_type'] == col\n    m2 = df_de_train['sm_name'] == drug\n    m2 = m2 & m\n    plt.plot( X[m2].values.ravel(), label = col )\n    plt.title(drug   ,fontsize = 20)\n    plt.legend(fontsize = 20 )\n    \n    \n    plt.subplot(1,2,2)\n    #print(drug)\n    \n    #plt.subplot(1,2,2)\n    col = 'B cells'\n    m = df_de_train['cell_type'] == col\n    m2 = df_de_train['sm_name'] == drug\n    m2 = m2 & m\n    plt.plot( X[m2].values.ravel(), label = col )\n    plt.title(drug   ,fontsize = 20)\n    plt.legend(fontsize = 20 )\n\n    col = 'Myeloid cells'\n    m = df_de_train['cell_type'] == col\n    #plt.subplot(1,2,1)\n    m2 = df_de_train['sm_name'] == drug\n    m2 = m2 & m\n    plt.plot( X[m2].values.ravel() , label = col )\n    plt.title(drug + ' ' + col  ,fontsize = 20)\n    plt.legend(fontsize = 20 )\n    \n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.835306Z","iopub.execute_input":"2023-11-30T19:12:33.835565Z","iopub.status.idle":"2023-11-30T19:12:46.599107Z","shell.execute_reply.started":"2023-11-30T19:12:33.83554Z","shell.execute_reply":"2023-11-30T19:12:46.598395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['cell_type'] == 'NK cells'\nm.sum()\ndrug = 'Crizotinib'\nm2 = df_de_train['sm_name'] == drug\n(m & m2).sum()\nX[m&m2]","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:46.602723Z","iopub.execute_input":"2023-11-30T19:12:46.603233Z","iopub.status.idle":"2023-11-30T19:12:46.623364Z","shell.execute_reply.started":"2023-11-30T19:12:46.603187Z","shell.execute_reply":"2023-11-30T19:12:46.622637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drugs higher effect on myeloids:\nl6 = ['Crizotinib',\n'Dabrafenib',\n'R428',\n'Porcn Inhibitor III',\n'Foretinib',\n'Dactolisib']\n\nfor ct in [ 'B cells', 'Myeloid cells']:\n    m = df_de_train['cell_type'] == ct\n    print( m.sum() )\n    cm = X[m].T.corr() \n    ll = df_de_train[m]['sm_name']\n    dd = {True : '   Higher on Myeloids  ', False: ''}\n    ll = [t +' '+ str( dd[t in l6] ) for t in ll]\n    cm.index = cm.columns = ll\n\n    sns.clustermap( cm.round(2), annot= True)\n    plt.title(ct, fontsize = 20 )\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:46.624333Z","iopub.execute_input":"2023-11-30T19:12:46.624545Z","iopub.status.idle":"2023-11-30T19:12:48.647559Z","shell.execute_reply.started":"2023-11-30T19:12:46.624525Z","shell.execute_reply":"2023-11-30T19:12:48.646613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor ct in [ 'NK cells']:#, 'Myeloid cells']:\n    m = df_de_train['cell_type'] == ct\n    print( m.sum() )\n    cm = X[m].T.corr() \n    cm.index = cm.columns = df_de_train[m]['sm_name']\n\n    clustergrid = sns.clustermap( cm.round(2), annot= False)\n    \n    reordered_columns = clustergrid.dendrogram_col.reordered_ind\n    reordered_rows = clustergrid.dendrogram_row.reordered_ind\n    print(len(reordered_rows), len(reordered_columns))\n    print(list(cm.index[reordered_rows]))\n\n    plt.title(ct, fontsize = 20 )\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:48.648863Z","iopub.execute_input":"2023-11-30T19:12:48.649374Z","iopub.status.idle":"2023-11-30T19:12:50.154829Z","shell.execute_reply.started":"2023-11-30T19:12:48.64935Z","shell.execute_reply":"2023-11-30T19:12:50.153903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor ct in [ 'NK cells']:#, 'Myeloid cells']:\n    m = df_de_train['cell_type'] == ct\n    print( m.sum() )\n    cm = X[m].T.corr() \n    cm.index = cm.columns = df_de_train[m]['sm_name']\n\n    sns.clustermap( cm.round(2), annot= True)\n    plt.title(ct, fontsize = 20 )\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:50.156055Z","iopub.execute_input":"2023-11-30T19:12:50.15633Z","iopub.status.idle":"2023-11-30T19:13:24.109644Z","shell.execute_reply.started":"2023-11-30T19:12:50.156308Z","shell.execute_reply":"2023-11-30T19:13:24.108769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"public_ids = [\n    'LSM-43216', 'LSM-1050', 'LSM-45849', 'LSM-42800', 'LSM-1131', 'LSM-6335', 'LSM-1211',\n    'LSM-45239', 'LSM-1130', 'LSM-45786', 'LSM-5199', 'LSM-45281',\n    'LSM-6324', # 'ACY-1215' -> 'Ricolinostat'\n    'LSM-3309', 'LSM-1056', 'LSM-45591', 'LSM-46203', 'LSM-5662',\n    'LSM-47134',  # 'SB-2342' -> '5-(9-Isopropyl-8-methyl-2-morpholino-9H-purin-6-yl)pyrimidin-2-amine  '\n    'LSM-45637', 'LSM-1127', 'LSM-46971', 'LSM-1172', 'LSM-46042', 'LSM-1101', 'LSM-45758',\n    'LSM-5218', 'LSM-2287', 'LSM-1014',\n    'LSM-1040', #  'fostamatinib' -> 'Tamatinib'\n    'LSM-1476;LSM-5290',\n    'LSM-45680',  # 'basimglurant' -> 'RG7090'\n    'LSM-4349',  # '5-iodotubercidin' -> 'IN1451'\n    'LSM-3425', 'LSM-45806',\n    'LSM-45616',  # 'SB-683698' -> 'TR-14035'\n    'LSM-1055',\n    'LSM-43281',  # 'C-646' -> 'STK219801'\n    'LSM-5690', 'LSM-1155', 'LSM-2499',\n    'LSM-2382',  # 'JTC-801' -> 'UNII-BXU45ZH6LI'\n    'LSM-45220', 'LSM-1037', 'LSM-1005', 'LSM-1180', 'LSM-36812',\n    'LSM-45924',  # 'filgotinib' -> 'GLPG0634'\n    'LSM-2013',  # 'TL-HRAS-61' -> TL_HRAS26'\n    'LSM-4738'\n]\nlist17 = ['Idelalisib', 'Crizotinib', 'Linagliptin', 'Palbociclib',\n       'Dabrafenib', 'Alvocidib', 'LDN 193189', 'R428',\n       'Porcn Inhibitor III', 'Belinostat', 'Foretinib', 'MLN 2238',\n       'Penfluridol', 'Dactolisib', 'O-Demethylated Adapalene',\n       'Oprozomib (ONX 0912)', 'CHIR-99021']\nprint(len(list17))\nprint(len(public_ids))\nm = df_de_train['sm_lincs_id'].isin(public_ids )\npublic_names = list( df_de_train[m]['sm_name'].unique() )\nlist_publ = public_names\nprint('public_names', public_names)\nm =( ~df_de_train['sm_lincs_id'].isin(public_ids )) & ( ~df_de_train['sm_name'].isin(list17 ) )\nprint(m.sum() )\nlist_priv = list( df_de_train['sm_name'][m].unique() )\nprint(len(list_priv))\nprint('list_priv', list_priv )","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:13:24.110647Z","iopub.execute_input":"2023-11-30T19:13:24.110905Z","iopub.status.idle":"2023-11-30T19:13:24.133531Z","shell.execute_reply.started":"2023-11-30T19:13:24.110882Z","shell.execute_reply":"2023-11-30T19:13:24.132619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['sm_name'].isin( list_priv)\nm = m & (df_de_train['cell_type'] == 'NK cells' )\ncm = X[m].T.corr()\ncm.index = cm.columns = df_de_train[m]['sm_name']\nclustergrid = sns.clustermap( cm.round(2), annot= False)\n\nreordered_columns = clustergrid.dendrogram_col.reordered_ind\nreordered_rows = clustergrid.dendrogram_row.reordered_ind\nprint(len(reordered_rows), len(reordered_columns))\nprint(list(cm.index[reordered_rows]))\n\nplt.show()\nprint(list(cm.index[reordered_rows]))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:13:24.136114Z","iopub.execute_input":"2023-11-30T19:13:24.136899Z","iopub.status.idle":"2023-11-30T19:13:25.154964Z","shell.execute_reply.started":"2023-11-30T19:13:24.13687Z","shell.execute_reply":"2023-11-30T19:13:25.154076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['sm_name'].isin( list_priv)\nm = m | df_de_train['sm_name'].isin( list17)\nm = m & (df_de_train['cell_type'] == 'NK cells' )\ncm = X[m].T.corr()\n\nll = df_de_train[m]['sm_name']\ndd = {True : '   Higher on Myeloids  ', False: ''}\nll = [t +' '+ str( dd[t in l6] ) for t in ll]\n\ncm.index = cm.columns = ll\nclustergrid = sns.clustermap( cm.round(2), annot= False, cmap = 'coolwarm')\n\nreordered_columns = clustergrid.dendrogram_col.reordered_ind\nreordered_rows = clustergrid.dendrogram_row.reordered_ind\nprint(len(reordered_rows), len(reordered_columns))\nprint(list(cm.index[reordered_rows]))\n\nplt.show()\nprint(list(cm.index[reordered_rows]))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:13:25.156349Z","iopub.execute_input":"2023-11-30T19:13:25.156856Z","iopub.status.idle":"2023-11-30T19:13:27.639068Z","shell.execute_reply.started":"2023-11-30T19:13:25.15683Z","shell.execute_reply":"2023-11-30T19:13:27.638248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm2 = cm.copy()\ncm2.columns =  df_de_train[m]['sm_name']\ncm2 = cm2[l6]\ncm2 = cm2[~cm2.index.isin(list17)]\ncm2['Mean'] = cm2.mean(axis = 1)\ncm2['Mean'].sort_values(ascending = False ).head(30)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:13:27.640083Z","iopub.execute_input":"2023-11-30T19:13:27.640835Z","iopub.status.idle":"2023-11-30T19:13:27.65656Z","shell.execute_reply.started":"2023-11-30T19:13:27.64081Z","shell.execute_reply":"2023-11-30T19:13:27.655683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['sm_name'].isin( list_publ)\nm = m | df_de_train['sm_name'].isin( list17)\nm = m & (df_de_train['cell_type'] == 'NK cells' )\ncm = X[m].T.corr()\n\nll = df_de_train[m]['sm_name']\ndd = {True : '   Higher on Myeloids  ', False: ''}\nll = [t +' '+ str( dd[t in l6] ) for t in ll]\n\ncm.index = cm.columns = ll\nclustergrid = sns.clustermap( cm.round(2), annot= False, cmap = 'coolwarm')\n\nreordered_columns = clustergrid.dendrogram_col.reordered_ind\nreordered_rows = clustergrid.dendrogram_row.reordered_ind\nprint(len(reordered_rows), len(reordered_columns))\nprint(list(cm.index[reordered_rows]))\n\nplt.show()\nprint(list(cm.index[reordered_rows]))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:14:06.844517Z","iopub.execute_input":"2023-11-30T19:14:06.844852Z","iopub.status.idle":"2023-11-30T19:14:07.738467Z","shell.execute_reply.started":"2023-11-30T19:14:06.844825Z","shell.execute_reply":"2023-11-30T19:14:07.737721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm2 = cm.copy()\ncm2.columns =  df_de_train[m]['sm_name']\ncm2 = cm2[l6]\ncm2 = cm2[~cm2.index.isin(list17)]\ncm2['Mean'] = cm2.mean(axis = 1)\ncm2['Mean'].sort_values(ascending = False ).head(30)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:14:08.82708Z","iopub.execute_input":"2023-11-30T19:14:08.827837Z","iopub.status.idle":"2023-11-30T19:14:08.843042Z","shell.execute_reply.started":"2023-11-30T19:14:08.827813Z","shell.execute_reply":"2023-11-30T19:14:08.841948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm2.index","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:15:01.960768Z","iopub.execute_input":"2023-11-30T19:15:01.961087Z","iopub.status.idle":"2023-11-30T19:15:01.967543Z","shell.execute_reply.started":"2023-11-30T19:15:01.961065Z","shell.execute_reply":"2023-11-30T19:15:01.96653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final timing","metadata":{}},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )\nprint('%.1f minutes passed total '%( (time.time()-t0start)/60)  )\nprint('%.2f hours passed total '%( (time.time()-t0start)/3600)  )","metadata":{},"execution_count":null,"outputs":[]}]}