{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about \n\nEDA for Single Cell Perturbations - mainly after the end of the challenge, some at the late end.\n\nIncludes analysis of some findings by the others: AmbrosM, Antoine Passemiers and Jalil Nourisa...\n\n","metadata":{}},{"cell_type":"markdown","source":"## Preliminaries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport time\nt0start = time.time() \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport pandas as pd\nimport tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-19T13:25:14.571822Z","iopub.execute_input":"2023-12-19T13:25:14.572985Z","iopub.status.idle":"2023-12-19T13:25:14.582534Z","shell.execute_reply.started":"2023-12-19T13:25:14.572942Z","shell.execute_reply":"2023-12-19T13:25:14.581183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_de_train = pd.read_parquet(fn)# , index_col = 0)\nprint(df_de_train.shape)\ndisplay(df_de_train )\n\nplt.figure(figsize = (20,4) )\nv = df_de_train.iloc[:,5:].max(axis = 0 ).sort_values(ascending = False, key = abs )\nplt.plot(v.values,'*-')\nplt.title('Max-abs DE for genes',fontsize = 20 )\nplt.grid()\nplt.show()\ndisplay(v.head(15))\n\n# %%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_id_map = pd.read_csv(fn,index_col = 0)\nprint(df_id_map.shape)\ndisplay(df_id_map)\nfn = '/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'\ndf_sample_submit = pd.read_csv(fn, index_col = 0)\nprint(df_sample_submit.shape)\ndisplay( df_sample_submit )\n\n\nfeatures_columns = ['cell_type','sm_name'] # \n# selected_features_columns = ['cell_type','sm_name']  # [ 'sm_name'] # What features to consider - others not used \n#     # For OP2 task we have two features - cell-type and compound - both categorical\n#     # Some simple models (like Ridge) better work when one uses only one feature - 'sm_name' (=\"compound\") - so you can try that option. \n#     # But seems pyboost is powerful enough to proper work with the both features.\n#     # Table with results for Ridge can be found here: https://www.kaggle.com/code/alexandervc/op2-target-encoders?scriptVersionId=148576642&cellId=1\n#     # Or here: https://docs.google.com/spreadsheets/d/1APN63PMaWZygVjYimK9Ivt0RvifdAU5JRYkxiDn4szw/edit?usp=sharing (sheet \"Encoders\") and discussed here:\n    \nX_submit_categorical = df_id_map[features_columns] # selected_features_columns] # pd.DataFrame(df_id_map, columns= features_columns )\nX_full_categorical = df_de_train[features_columns]# selected_features_columns ]\n\nY_full = df_de_train.iloc[:,5:].values\nprint('X_submit_categorical.shape, X_full_categorical.shape ,  Y_full.shape', X_submit_categorical.shape, X_full_categorical.shape ,  Y_full.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:25:15.795139Z","iopub.execute_input":"2023-12-19T13:25:15.796142Z","iopub.status.idle":"2023-12-19T13:25:22.315083Z","shell.execute_reply.started":"2023-12-19T13:25:15.796105Z","shell.execute_reply":"2023-12-19T13:25:22.313802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Strange bi-modality of genes de-p-values (after AmbrosM)\n\nAmbrosM https://www.kaggle.com/competitions/open-problems-single-cell-perturbations/discussion/458661 \npointed out and commented on stange bimodality for input de_train data   for some samples.\n\n\nAmbrosM suggests that the additional \"bell\" is artifact of Limma. Which comes from those genes which were not expressed at all.\n(Quote: \"The values for the genes which are not expressed (orange) have a distribution with an unusual shape, and it is strange that positive differential expressions are reported when not a single piece of RNA is counted.\")\n\nIt can be seen that lots of genes have that stange phenomena.\n\n\n","metadata":{}},{"cell_type":"markdown","source":"## 'Scriptaid', 'Foretinib'","metadata":{}},{"cell_type":"code","source":"%%time\nfor drug in ['Scriptaid', 'Foretinib']:#  ['Clotrimazole', 'Mometasone Furoate', 'Idelalisib', 'Vandetanib', 'Bosutinib', 'Ceritinib', 'Lamivudine', 'Crizotinib', 'Cabozantinib', 'Flutamide', 'Dasatinib', 'Selumetinib', 'Trametinib', 'ABT-199 (GDC-0199)', 'Oxybenzone', 'Vorinostat', 'Raloxifene', 'Linagliptin', 'Lapatinib', 'Canertinib', 'Disulfiram' ]:\n    m = df_de_train['sm_name'] == drug\n    print(m.sum() )\n    fig = plt.figure(figsize = (20,5));c = 0\n    plt.suptitle(drug, fontsize  = 20 )\n    n_x_subplots = m.sum()\n    for  index, row in  df_de_train[m].iloc[:,5:].iterrows():\n        c += 1; fig.add_subplot(1,n_x_subplots ,c)\n    \n        \n        ct = df_de_train.loc[index, 'cell_type']\n        print(index, ct)\n        print(row.value_counts().head(4))\n        plt.hist(row, bins = 100)\n        #plt.title(drug + ' ' + ct , fontsize = 20)\n        plt.title(ct , fontsize = 20)\n    plt.show()\n\n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-19T14:00:42.529333Z","iopub.execute_input":"2023-12-19T14:00:42.529838Z","iopub.status.idle":"2023-12-19T14:00:45.803618Z","shell.execute_reply.started":"2023-12-19T14:00:42.529802Z","shell.execute_reply":"2023-12-19T14:00:45.802734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print( list(df_de_train['sm_name'].unique()) )","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:53:47.886326Z","iopub.execute_input":"2023-12-19T13:53:47.887864Z","iopub.status.idle":"2023-12-19T13:53:47.897063Z","shell.execute_reply.started":"2023-12-19T13:53:47.887793Z","shell.execute_reply":"2023-12-19T13:53:47.894920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor drug in ['Clotrimazole', 'Mometasone Furoate', 'Idelalisib', 'Vandetanib', 'Bosutinib', 'Ceritinib', 'Lamivudine', 'Crizotinib', 'Cabozantinib', 'Flutamide', 'Dasatinib', 'Selumetinib', 'Trametinib', 'ABT-199 (GDC-0199)', 'Oxybenzone', 'Vorinostat', 'Raloxifene', 'Linagliptin', 'Lapatinib', 'Canertinib', 'Disulfiram' ]:\n    m = df_de_train['sm_name'] == drug\n    print(m.sum() )\n    fig = plt.figure(figsize = (20,5));c = 0\n    plt.suptitle(drug, fontsize  = 20 )\n    n_x_subplots = m.sum()\n    for  index, row in  df_de_train[m].iloc[:,5:].iterrows():\n        c += 1; fig.add_subplot(1,n_x_subplots ,c)\n    \n        \n        ct = df_de_train.loc[index, 'cell_type']\n        print(index, ct)\n#         print(row.value_counts().head(4))\n        plt.hist(row, bins = 100)\n        #plt.title(drug + ' ' + ct , fontsize = 20)\n        plt.title(ct , fontsize = 20)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-12-19T13:58:53.298644Z","iopub.execute_input":"2023-12-19T13:58:53.299091Z","iopub.status.idle":"2023-12-19T13:59:25.902350Z","shell.execute_reply.started":"2023-12-19T13:58:53.299057Z","shell.execute_reply":"2023-12-19T13:59:25.901270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Each(!) gene is stongly changed under at least one perturbation. \"Each\" - is strange","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = (20,4) )\nv = df_de_train.iloc[:,5:].abs().max(axis = 0 ).sort_values(ascending = False, key = abs )\nplt.plot(v.values,'*-')\nplt.title('Max-abs DE for genes',fontsize = 20 )\nplt.grid()\nplt.show()\ndisplay(v.head(15))\ndisplay(v.tail(15))\n","metadata":{"execution":{"iopub.status.busy":"2023-12-19T08:06:59.900536Z","iopub.execute_input":"2023-12-19T08:06:59.900917Z","iopub.status.idle":"2023-12-19T08:07:00.465914Z","shell.execute_reply.started":"2023-12-19T08:06:59.900884Z","shell.execute_reply":"2023-12-19T08:07:00.464622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"v = df_de_train.iloc[:,5:].abs().max(axis = 0 )\ndisplay(v.sort_values().head(10))\nprint('Min (log10(p-val)):',v.min() )\n\nplt.figure(figsize = (20,3))\nplt.hist(v,bins = 1000)\nplt.grid()\nplt.title('Maximum (across samples=perturbations) change(p-val) for genes. \\n Strange that for all genes it is quite big  ',fontsize = 20)\nplt.show()\nprint(10**(-np.median(v)))\n\nfor q in [0.99,0.95,0.9,0.7,0.5]:\n    v = df_de_train.iloc[:,5:].abs().quantile(q=q, axis = 0 )\n    # display(v.sort_values().head(10))\n    # print('Min (log10(p-val)):',v.min() )\n    plt.figure(figsize = (20,3))\n    plt.hist(v,bins = 1000)\n    plt.grid()\n    plt.title('quantile ' +str(q)+' (across samples=perturbations) change(p-val) for genes.  ',fontsize = 20)\n    plt.show()\n    print(10**(-np.median(v)))","metadata":{"execution":{"iopub.status.busy":"2023-12-19T08:07:00.468300Z","iopub.execute_input":"2023-12-19T08:07:00.468719Z","iopub.status.idle":"2023-12-19T08:07:15.641354Z","shell.execute_reply.started":"2023-12-19T08:07:00.468682Z","shell.execute_reply":"2023-12-19T08:07:15.640292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# There 17 perturbations where MORE THAN a HALF genes genes stronger than 0.001 p-value - strange","metadata":{}},{"cell_type":"code","source":"q = 0.5\nv = df_de_train.iloc[:,5:].abs().quantile(q=q, axis = 1)\nNN = 25\nprint('count samples with quantile ',q, 'less than 0.001' , (v>3).sum() )\nt = df_de_train.iloc[v.sort_values().tail(NN).index,:2]\nt['quantile'] = v.sort_values().tail(NN).values\ndisplay(t)\nplt.figure(figsize = (20,3))\nplt.hist(v,bins = 1000)\nplt.grid()\nplt.title('quantile '+str(q)+' (across genes) change( = log10-pval) for genes. \\n Strange that there are samples with very high quantile  ',fontsize = 20)\nplt.show()\nprint(10**(-np.median(v)))","metadata":{"execution":{"iopub.status.busy":"2023-12-19T08:08:54.533094Z","iopub.execute_input":"2023-12-19T08:08:54.533529Z","iopub.status.idle":"2023-12-19T08:08:56.897754Z","shell.execute_reply.started":"2023-12-19T08:08:54.533490Z","shell.execute_reply":"2023-12-19T08:08:56.896646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor q in [0.7,0.9,0.95]:\n    v = df_de_train.iloc[:,5:].abs().quantile(q=q, axis = 1)\n    print('count samples with quantile ',q, 'less than 0.001' , (v>3).sum() )\n    NN = 25\n    t = df_de_train.iloc[v.sort_values().tail(NN).index,:2]\n    t['quantile'] = v.sort_values().tail(NN).values\n    display(t)\n    plt.figure(figsize = (20,3))\n    plt.hist(v,bins = 1000)\n    plt.grid()\n    plt.title('quantile '+str(q)+' (across genes) change( = log10-pval) for genes. \\n Strange that there are samples with very high quantile  ',fontsize = 20)\n    plt.show()\n    print(10**(-np.median(v)))","metadata":{"execution":{"iopub.status.busy":"2023-12-19T08:18:49.873078Z","iopub.execute_input":"2023-12-19T08:18:49.873529Z","iopub.status.idle":"2023-12-19T08:18:57.784259Z","shell.execute_reply.started":"2023-12-19T08:18:49.873495Z","shell.execute_reply":"2023-12-19T08:18:57.782865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Uniform distribution under null hypothesis\n\nBased on   https://www.kaggle.com/competitions/open-problems-single-cell-perturbations/discussion/461159\n\nAntoine Passemiers, JALIL NOURISA · 20TH IN THIS COMPETITION · POSTED 5 DAYS AGO\n","metadata":{}},{"cell_type":"code","source":"X = df_de_train.iloc[:,5:].values\nv = X.ravel()\nv = 10**(-np.abs(v))\nplt.hist(v,bins = 300)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T21:06:23.017377Z","iopub.execute_input":"2023-12-17T21:06:23.017713Z","iopub.status.idle":"2023-12-17T21:06:23.825578Z","shell.execute_reply.started":"2023-12-17T21:06:23.017687Z","shell.execute_reply":"2023-12-17T21:06:23.824113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for uv in df_de_train['cell_type'].unique():\n    m = df_de_train['cell_type'] == uv \n    X = df_de_train[m].iloc[:,5:].values\n    v = X.ravel()\n    v = 10**(-np.abs(v))\n    print(uv, m.sum(),  np.min(v))\n    plt.hist(v,bins = 100)\n    plt.title(uv, fontsize = 20 )\n    plt.show()\n    plt.hist(v[v<1e-5],bins = 100)\n    plt.title(uv, fontsize = 20 )\n    plt.show()    ","metadata":{"execution":{"iopub.status.busy":"2023-12-17T22:20:06.833688Z","iopub.execute_input":"2023-12-17T22:20:06.834044Z","iopub.status.idle":"2023-12-17T22:20:11.229819Z","shell.execute_reply.started":"2023-12-17T22:20:06.834015Z","shell.execute_reply":"2023-12-17T22:20:11.228967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_de_train['cell_type'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-12-17T21:03:56.660706Z","iopub.execute_input":"2023-12-17T21:03:56.661019Z","iopub.status.idle":"2023-12-17T21:03:56.669931Z","shell.execute_reply.started":"2023-12-17T21:03:56.660990Z","shell.execute_reply":"2023-12-17T21:03:56.668927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_de_train.iloc[:,5:]\nX.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.240654Z","iopub.execute_input":"2023-11-30T19:12:33.240913Z","iopub.status.idle":"2023-11-30T19:12:33.272957Z","shell.execute_reply.started":"2023-11-30T19:12:33.240893Z","shell.execute_reply":"2023-11-30T19:12:33.272095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['cell_type'] ==  'Myeloid cells'#  'B cells'\nv = X[m].values.ravel()\ns1 = pd.Series(v).describe()\ns1","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.274035Z","iopub.execute_input":"2023-11-30T19:12:33.274550Z","iopub.status.idle":"2023-11-30T19:12:33.301344Z","shell.execute_reply.started":"2023-11-30T19:12:33.274528Z","shell.execute_reply":"2023-11-30T19:12:33.300576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_df = []\nfor col in ['B cells' , 'Myeloid cells' ]:\n    m = df_de_train['cell_type'] ==  col\n    v = X[m].values.ravel()\n    s2 = pd.Series(v).describe()\n    s2.name = col\n    list_df.append( s2.to_frame() ) \nd = pd.concat( list_df, axis = 1)\nd","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.302172Z","iopub.execute_input":"2023-11-30T19:12:33.302399Z","iopub.status.idle":"2023-11-30T19:12:33.339917Z","shell.execute_reply.started":"2023-11-30T19:12:33.302381Z","shell.execute_reply":"2023-11-30T19:12:33.339180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in ['B cells' , 'Myeloid cells' ]:\n    m = df_de_train['cell_type'] ==  col\n    v = X[m].values.ravel()\n    plt.plot(v)\n    plt.title(col,fontsize = 20)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.340566Z","iopub.execute_input":"2023-11-30T19:12:33.340776Z","iopub.status.idle":"2023-11-30T19:12:33.823210Z","shell.execute_reply.started":"2023-11-30T19:12:33.340756Z","shell.execute_reply":"2023-11-30T19:12:33.822382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['cell_type'] == 'Myeloid cells' \ndf_de_train[m]['sm_name'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.824860Z","iopub.execute_input":"2023-11-30T19:12:33.825272Z","iopub.status.idle":"2023-11-30T19:12:33.834329Z","shell.execute_reply.started":"2023-11-30T19:12:33.825246Z","shell.execute_reply":"2023-11-30T19:12:33.833531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Got from that:\n# Drugs higher effect on myeloids:\nl6 = ['Crizotinib',\n'Dabrafenib',\n'R428',\n'Porcn Inhibitor III',\n'Foretinib',\n'Dactolisib']\n\n\nprint( col  )\n\nm = df_de_train['cell_type'] == 'Myeloid cells' \ni = 0;\nfor drug in df_de_train['sm_name'][m].unique(): \n    plt.figure(figsize = (20,3) )\n    plt.subplot(1,2,1)\n    i+=1\n    print(i, drug)\n    col = 'Myeloid cells'\n    m = df_de_train['cell_type'] == col\n    #plt.subplot(1,2,1)\n    m2 = df_de_train['sm_name'] == drug\n    m2 = m2 & m\n    plt.plot( X[m2].values.ravel() , label = col )\n    plt.title(drug + ' ' + col  ,fontsize = 20)\n    \n    #plt.subplot(1,2,2)\n    col = 'B cells'\n    m = df_de_train['cell_type'] == col\n    m2 = df_de_train['sm_name'] == drug\n    m2 = m2 & m\n    plt.plot( X[m2].values.ravel(), label = col )\n    plt.title(drug   ,fontsize = 20)\n    plt.legend(fontsize = 20 )\n    \n    \n    plt.subplot(1,2,2)\n    #print(drug)\n    \n    #plt.subplot(1,2,2)\n    col = 'B cells'\n    m = df_de_train['cell_type'] == col\n    m2 = df_de_train['sm_name'] == drug\n    m2 = m2 & m\n    plt.plot( X[m2].values.ravel(), label = col )\n    plt.title(drug   ,fontsize = 20)\n    plt.legend(fontsize = 20 )\n\n    col = 'Myeloid cells'\n    m = df_de_train['cell_type'] == col\n    #plt.subplot(1,2,1)\n    m2 = df_de_train['sm_name'] == drug\n    m2 = m2 & m\n    plt.plot( X[m2].values.ravel() , label = col )\n    plt.title(drug + ' ' + col  ,fontsize = 20)\n    plt.legend(fontsize = 20 )\n    \n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:33.835306Z","iopub.execute_input":"2023-11-30T19:12:33.835565Z","iopub.status.idle":"2023-11-30T19:12:46.599107Z","shell.execute_reply.started":"2023-11-30T19:12:33.835540Z","shell.execute_reply":"2023-11-30T19:12:46.598395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['cell_type'] == 'NK cells'\nm.sum()\ndrug = 'Crizotinib'\nm2 = df_de_train['sm_name'] == drug\n(m & m2).sum()\nX[m&m2]","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:46.602723Z","iopub.execute_input":"2023-11-30T19:12:46.603233Z","iopub.status.idle":"2023-11-30T19:12:46.623364Z","shell.execute_reply.started":"2023-11-30T19:12:46.603187Z","shell.execute_reply":"2023-11-30T19:12:46.622637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drugs higher effect on myeloids:\nl6 = ['Crizotinib',\n'Dabrafenib',\n'R428',\n'Porcn Inhibitor III',\n'Foretinib',\n'Dactolisib']\n\nfor ct in [ 'B cells', 'Myeloid cells']:\n    m = df_de_train['cell_type'] == ct\n    print( m.sum() )\n    cm = X[m].T.corr() \n    ll = df_de_train[m]['sm_name']\n    dd = {True : '   Higher on Myeloids  ', False: ''}\n    ll = [t +' '+ str( dd[t in l6] ) for t in ll]\n    cm.index = cm.columns = ll\n\n    sns.clustermap( cm.round(2), annot= True)\n    plt.title(ct, fontsize = 20 )\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:46.624333Z","iopub.execute_input":"2023-11-30T19:12:46.624545Z","iopub.status.idle":"2023-11-30T19:12:48.647559Z","shell.execute_reply.started":"2023-11-30T19:12:46.624525Z","shell.execute_reply":"2023-11-30T19:12:48.646613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor ct in [ 'NK cells']:#, 'Myeloid cells']:\n    m = df_de_train['cell_type'] == ct\n    print( m.sum() )\n    cm = X[m].T.corr() \n    cm.index = cm.columns = df_de_train[m]['sm_name']\n\n    clustergrid = sns.clustermap( cm.round(2), annot= False)\n    \n    reordered_columns = clustergrid.dendrogram_col.reordered_ind\n    reordered_rows = clustergrid.dendrogram_row.reordered_ind\n    print(len(reordered_rows), len(reordered_columns))\n    print(list(cm.index[reordered_rows]))\n\n    plt.title(ct, fontsize = 20 )\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:48.648863Z","iopub.execute_input":"2023-11-30T19:12:48.649374Z","iopub.status.idle":"2023-11-30T19:12:50.154829Z","shell.execute_reply.started":"2023-11-30T19:12:48.649350Z","shell.execute_reply":"2023-11-30T19:12:50.153903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor ct in [ 'NK cells']:#, 'Myeloid cells']:\n    m = df_de_train['cell_type'] == ct\n    print( m.sum() )\n    cm = X[m].T.corr() \n    cm.index = cm.columns = df_de_train[m]['sm_name']\n\n    sns.clustermap( cm.round(2), annot= True)\n    plt.title(ct, fontsize = 20 )\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:12:50.156055Z","iopub.execute_input":"2023-11-30T19:12:50.156330Z","iopub.status.idle":"2023-11-30T19:13:24.109644Z","shell.execute_reply.started":"2023-11-30T19:12:50.156308Z","shell.execute_reply":"2023-11-30T19:13:24.108769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"public_ids = [\n    'LSM-43216', 'LSM-1050', 'LSM-45849', 'LSM-42800', 'LSM-1131', 'LSM-6335', 'LSM-1211',\n    'LSM-45239', 'LSM-1130', 'LSM-45786', 'LSM-5199', 'LSM-45281',\n    'LSM-6324', # 'ACY-1215' -> 'Ricolinostat'\n    'LSM-3309', 'LSM-1056', 'LSM-45591', 'LSM-46203', 'LSM-5662',\n    'LSM-47134',  # 'SB-2342' -> '5-(9-Isopropyl-8-methyl-2-morpholino-9H-purin-6-yl)pyrimidin-2-amine  '\n    'LSM-45637', 'LSM-1127', 'LSM-46971', 'LSM-1172', 'LSM-46042', 'LSM-1101', 'LSM-45758',\n    'LSM-5218', 'LSM-2287', 'LSM-1014',\n    'LSM-1040', #  'fostamatinib' -> 'Tamatinib'\n    'LSM-1476;LSM-5290',\n    'LSM-45680',  # 'basimglurant' -> 'RG7090'\n    'LSM-4349',  # '5-iodotubercidin' -> 'IN1451'\n    'LSM-3425', 'LSM-45806',\n    'LSM-45616',  # 'SB-683698' -> 'TR-14035'\n    'LSM-1055',\n    'LSM-43281',  # 'C-646' -> 'STK219801'\n    'LSM-5690', 'LSM-1155', 'LSM-2499',\n    'LSM-2382',  # 'JTC-801' -> 'UNII-BXU45ZH6LI'\n    'LSM-45220', 'LSM-1037', 'LSM-1005', 'LSM-1180', 'LSM-36812',\n    'LSM-45924',  # 'filgotinib' -> 'GLPG0634'\n    'LSM-2013',  # 'TL-HRAS-61' -> TL_HRAS26'\n    'LSM-4738'\n]\nlist17 = ['Idelalisib', 'Crizotinib', 'Linagliptin', 'Palbociclib',\n       'Dabrafenib', 'Alvocidib', 'LDN 193189', 'R428',\n       'Porcn Inhibitor III', 'Belinostat', 'Foretinib', 'MLN 2238',\n       'Penfluridol', 'Dactolisib', 'O-Demethylated Adapalene',\n       'Oprozomib (ONX 0912)', 'CHIR-99021']\nprint(len(list17))\nprint(len(public_ids))\nm = df_de_train['sm_lincs_id'].isin(public_ids )\npublic_names = list( df_de_train[m]['sm_name'].unique() )\nlist_publ = public_names\nprint('public_names', public_names)\nm =( ~df_de_train['sm_lincs_id'].isin(public_ids )) & ( ~df_de_train['sm_name'].isin(list17 ) )\nprint(m.sum() )\nlist_priv = list( df_de_train['sm_name'][m].unique() )\nprint(len(list_priv))\nprint('list_priv', list_priv )","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:13:24.110647Z","iopub.execute_input":"2023-11-30T19:13:24.110905Z","iopub.status.idle":"2023-11-30T19:13:24.133531Z","shell.execute_reply.started":"2023-11-30T19:13:24.110882Z","shell.execute_reply":"2023-11-30T19:13:24.132619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['sm_name'].isin( list_priv)\nm = m & (df_de_train['cell_type'] == 'NK cells' )\ncm = X[m].T.corr()\ncm.index = cm.columns = df_de_train[m]['sm_name']\nclustergrid = sns.clustermap( cm.round(2), annot= False)\n\nreordered_columns = clustergrid.dendrogram_col.reordered_ind\nreordered_rows = clustergrid.dendrogram_row.reordered_ind\nprint(len(reordered_rows), len(reordered_columns))\nprint(list(cm.index[reordered_rows]))\n\nplt.show()\nprint(list(cm.index[reordered_rows]))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:13:24.136114Z","iopub.execute_input":"2023-11-30T19:13:24.136899Z","iopub.status.idle":"2023-11-30T19:13:25.154964Z","shell.execute_reply.started":"2023-11-30T19:13:24.136870Z","shell.execute_reply":"2023-11-30T19:13:25.154076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['sm_name'].isin( list_priv)\nm = m | df_de_train['sm_name'].isin( list17)\nm = m & (df_de_train['cell_type'] == 'NK cells' )\ncm = X[m].T.corr()\n\nll = df_de_train[m]['sm_name']\ndd = {True : '   Higher on Myeloids  ', False: ''}\nll = [t +' '+ str( dd[t in l6] ) for t in ll]\n\ncm.index = cm.columns = ll\nclustergrid = sns.clustermap( cm.round(2), annot= False, cmap = 'coolwarm')\n\nreordered_columns = clustergrid.dendrogram_col.reordered_ind\nreordered_rows = clustergrid.dendrogram_row.reordered_ind\nprint(len(reordered_rows), len(reordered_columns))\nprint(list(cm.index[reordered_rows]))\n\nplt.show()\nprint(list(cm.index[reordered_rows]))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:13:25.156349Z","iopub.execute_input":"2023-11-30T19:13:25.156856Z","iopub.status.idle":"2023-11-30T19:13:27.639068Z","shell.execute_reply.started":"2023-11-30T19:13:25.156830Z","shell.execute_reply":"2023-11-30T19:13:27.638248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm2 = cm.copy()\ncm2.columns =  df_de_train[m]['sm_name']\ncm2 = cm2[l6]\ncm2 = cm2[~cm2.index.isin(list17)]\ncm2['Mean'] = cm2.mean(axis = 1)\ncm2['Mean'].sort_values(ascending = False ).head(30)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:13:27.640083Z","iopub.execute_input":"2023-11-30T19:13:27.640835Z","iopub.status.idle":"2023-11-30T19:13:27.656560Z","shell.execute_reply.started":"2023-11-30T19:13:27.640810Z","shell.execute_reply":"2023-11-30T19:13:27.655683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df_de_train['sm_name'].isin( list_publ)\nm = m | df_de_train['sm_name'].isin( list17)\nm = m & (df_de_train['cell_type'] == 'NK cells' )\ncm = X[m].T.corr()\n\nll = df_de_train[m]['sm_name']\ndd = {True : '   Higher on Myeloids  ', False: ''}\nll = [t +' '+ str( dd[t in l6] ) for t in ll]\n\ncm.index = cm.columns = ll\nclustergrid = sns.clustermap( cm.round(2), annot= False, cmap = 'coolwarm')\n\nreordered_columns = clustergrid.dendrogram_col.reordered_ind\nreordered_rows = clustergrid.dendrogram_row.reordered_ind\nprint(len(reordered_rows), len(reordered_columns))\nprint(list(cm.index[reordered_rows]))\n\nplt.show()\nprint(list(cm.index[reordered_rows]))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:14:06.844517Z","iopub.execute_input":"2023-11-30T19:14:06.844852Z","iopub.status.idle":"2023-11-30T19:14:07.738467Z","shell.execute_reply.started":"2023-11-30T19:14:06.844825Z","shell.execute_reply":"2023-11-30T19:14:07.737721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm2 = cm.copy()\ncm2.columns =  df_de_train[m]['sm_name']\ncm2 = cm2[l6]\ncm2 = cm2[~cm2.index.isin(list17)]\ncm2['Mean'] = cm2.mean(axis = 1)\ncm2['Mean'].sort_values(ascending = False ).head(30)","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:14:08.827080Z","iopub.execute_input":"2023-11-30T19:14:08.827837Z","iopub.status.idle":"2023-11-30T19:14:08.843042Z","shell.execute_reply.started":"2023-11-30T19:14:08.827813Z","shell.execute_reply":"2023-11-30T19:14:08.841948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm2.index","metadata":{"execution":{"iopub.status.busy":"2023-11-30T19:15:01.960768Z","iopub.execute_input":"2023-11-30T19:15:01.961087Z","iopub.status.idle":"2023-11-30T19:15:01.967543Z","shell.execute_reply.started":"2023-11-30T19:15:01.961065Z","shell.execute_reply":"2023-11-30T19:15:01.966530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final timing","metadata":{}},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )\nprint('%.1f minutes passed total '%( (time.time()-t0start)/60)  )\nprint('%.2f hours passed total '%( (time.time()-t0start)/3600)  )","metadata":{},"execution_count":null,"outputs":[]}]}