{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nWe work with several datasets by CITE-seq technology. \n\nHere look on ligand-receptor pairs of CD proteins and look what is the correlation between corresponding proteins.\nWe expect correlation to be quite high. \n\n","metadata":{}},{"cell_type":"markdown","source":"# Key params","metadata":{}},{"cell_type":"code","source":"N_rows2take = int( 2e4 ) # Work only with first rows of the data - to speed up, avoid RAM crashes , etc... \n\nwork_with_rna_data = 'Yes'# If No - we will not store/work with protein data - only rna -  For some fast analysis we need only protein data, which much more smaller \nwork_with_protein_data = 'Yes' \n\n\ndict_datasets2consider = {}\ndict_datasets2consider['NIPS2022'] = 'Yes'\ndict_datasets2consider['NIPS2021'] = 'Yes'\ndict_datasets2consider['Kaggle2302'] = 'Yes'\ndict_datasets2consider['GSE148127 Mouse Brain'] = 'Yes'# 'No'\ndict_datasets2consider['GSE139891 HGC-Bcells'] = 'Yes'#'No'\ndict_datasets2consider['GSE160766 MouseSpleen'] = 'Yes'#'No'\ndict_datasets2consider['GSE155673 PBMC Covid'] = 1 # Can be a number - it indicates how many subdatsets from GSE155673 will be taked \n                                                   # There 12 subdatsets  \n                                                   # For case \"Yes\" we will take just one the first subdaset \n                                                   # subdatasets sizes about:  3000-9000 cells \n    \n\n# -----------------------------------------------------------------\n# Quick params modification : \n# -----------------------------------------------------------------\n\nif 0:  # Quick yes only for the first N\n    N = 3\n    for ii,k in enumerate(dict_datasets2consider):\n        if ii<N:\n            dict_datasets2consider[k] = 'Yes'\n        else:\n            dict_datasets2consider[k] = 'No'\n            \n\nif 0:  # Quick yes for all \n    for k in dict_datasets2consider:\n        dict_datasets2consider[k] = 'Yes'\n","metadata":{"execution":{"iopub.status.busy":"2023-03-30T16:27:16.222011Z","iopub.execute_input":"2023-03-30T16:27:16.223789Z","iopub.status.idle":"2023-03-30T16:27:16.267713Z","shell.execute_reply.started":"2023-03-30T16:27:16.223714Z","shell.execute_reply":"2023-03-30T16:27:16.266399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparations","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport time \nt0start = time.time()\nimport gc\n","metadata":{"execution":{"iopub.status.busy":"2023-03-30T16:27:32.417558Z","iopub.execute_input":"2023-03-30T16:27:32.418046Z","iopub.status.idle":"2023-03-30T16:27:33.234583Z","shell.execute_reply.started":"2023-03-30T16:27:32.418006Z","shell.execute_reply":"2023-03-30T16:27:33.233161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if 'research-project-01' not in os.path.join(dirname, filename): \n            print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-30T16:27:37.507990Z","iopub.execute_input":"2023-03-30T16:27:37.509335Z","iopub.status.idle":"2023-03-30T16:27:37.711981Z","shell.execute_reply.started":"2023-03-30T16:27:37.509271Z","shell.execute_reply":"2023-03-30T16:27:37.709775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scanpy\nimport scanpy as sc","metadata":{"execution":{"iopub.status.busy":"2023-03-30T16:27:44.327782Z","iopub.execute_input":"2023-03-30T16:27:44.328284Z","iopub.status.idle":"2023-03-30T16:28:06.600720Z","shell.execute_reply.started":"2023-03-30T16:27:44.328243Z","shell.execute_reply":"2023-03-30T16:28:06.598632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"dict_df_prot = {}\ndict_df_rna = {}\ndict_df_meta = {}\n","metadata":{"execution":{"iopub.status.busy":"2023-03-30T16:28:06.604123Z","iopub.execute_input":"2023-03-30T16:28:06.604611Z","iopub.status.idle":"2023-03-30T16:28:06.611944Z","shell.execute_reply.started":"2023-03-30T16:28:06.604544Z","shell.execute_reply":"2023-03-30T16:28:06.610002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NIPS22-Kaggle ","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\n\nif dict_datasets2consider['NIPS2022'] == 'Yes':\n\n    key4dict = 'NIPS2022'\n    \n    DATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\n    if work_with_protein_data == 'Yes':\n        FP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\n        df = pd.read_hdf(FP_CITE_TRAIN_TARGETS, start=0, stop=N_rows2take, )\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2)) \n\n    # %%time\n    if work_with_rna_data == 'Yes':\n        filename_rna_data = '/kaggle/input/open-problems-multimodal/train_cite_inputs.h5'\n        df = pd.read_hdf(filename_rna_data, start=0, stop=N_rows2take,)\n        l = [t.split('_')[1] for t in df.columns]; df.columns = l \n        index_save = df.index\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n\n\n    fn = '/kaggle/input/open-problems-multimodal/metadata.csv'\n    df = pd.read_csv(fn, index_col = 0 );\n    mask = df.index.isin(index_save); df = df[mask]\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect() \n","metadata":{"execution":{"iopub.status.busy":"2023-03-30T16:28:44.420426Z","iopub.execute_input":"2023-03-30T16:28:44.421439Z","iopub.status.idle":"2023-03-30T16:29:04.523801Z","shell.execute_reply.started":"2023-03-30T16:28:44.421391Z","shell.execute_reply":"2023-03-30T16:29:04.522295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NIPS21","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\n\nif dict_datasets2consider['NIPS2021'] == 'Yes':\n    key4dict = 'NIPS2021'\n    fn = '/kaggle/input/citeseqscrnaseqproteins-challenge-neurips2021/GSE194122_openproblems_neurips2021_cite_BMMC_processed.h5ad'\n    adata = sc.read(fn)\n    adata.var_names_make_unique()\n    print(adata)\n    \n    if work_with_protein_data == 'Yes':\n        mask = adata.var['feature_types' ] == 'ADT'\n        X = adata[:N_rows2take,mask].X.todense()\n        df = pd.DataFrame(X, columns = list(adata.var['feature_types' ][mask].index)  )\n        l = [t.replace('-1','') for t in df.columns]\n        df.columns = l\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        print( 'Protein data shape:', df.shape)\n        dict_df_prot[key4dict] = df\n        display(df.head(2)) #df\n    \n    if work_with_rna_data == 'Yes':\n        mask = adata.var['feature_types' ] != 'ADT'\n        X = adata[:N_rows2take,mask].X.todense()\n        df = pd.DataFrame(X, columns = list(adata.var['feature_types' ][mask].index)  )\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n    \n    df = adata.obs.iloc[:N_rows2take,:]\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect()\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-03-30T16:29:23.409273Z","iopub.execute_input":"2023-03-30T16:29:23.410023Z","iopub.status.idle":"2023-03-30T16:29:42.728987Z","shell.execute_reply.started":"2023-03-30T16:29:23.409979Z","shell.execute_reply":"2023-03-30T16:29:42.726721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kaggle 2023 02 community competition \n\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nif dict_datasets2consider['Kaggle2302'] == 'Yes':\n    key4dict = 'Kaggle2302'\n\n    if work_with_protein_data == 'Yes':\n        df = pd.read_csv('/kaggle/input/machine-learning-challenge-2-prediction/training_set_adt.csv', index_col = 0).T # Train targets     \n        print('Original shape', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2)) \n\n    if work_with_rna_data == 'Yes':\n        df = pd.read_csv('/kaggle/input/machine-learning-challenge-2-prediction/training_set_rna.csv', index_col = 0).T    # Train features \n        print('Original shape', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df.iloc[:N_rows2take,:]\n        display(df.head(2)) \n\n\n    df = pd.DataFrame(index = df.index)\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df.iloc[:N_rows2take,:]\n    display(df.head(2)) \n    gc.collect()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-30T16:29:54.719196Z","iopub.execute_input":"2023-03-30T16:29:54.719726Z","iopub.status.idle":"2023-03-30T16:29:56.377985Z","shell.execute_reply.started":"2023-03-30T16:29:54.719679Z","shell.execute_reply":"2023-03-30T16:29:56.376439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE148127 mouse \n\nhttps://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE148127\n\nWe used CITE-seq (10x Gemoics-based) to profile and compare the transcriptomes and cell surface expression of a set of immune markers of naïve brain myeloid cells from young-adult (12 weeks old) wild type C57Bl/6 mice, aged (72 weeks old) wild type C57Bl/6 mice and young-adult and aged C57Bl/6 mice with antibiotics induced gut microbiota depletion. In one sequencing round, we sequenced a total of 12 different mouse brains (3 young control, 3 aged control, 3 young depleted gut microbiota (ABX) and 3 aged depleted gut microbiota (ABX)).\n\n\nWe created an antibody pool consisting of 31 different antibodies (CCR2/CD192, C117/c-kit, CD11b, CD11c, CD172a/SIRPα, CD25, CD3, CD38, CD4, CD44, CD45, CD45R/B220, CD86, CD8a, CD90.1, Cx3cr1, F4/80, I-A/I-E, Ly6C, Ly6G, NK1.1, PD-1, PD-L1, CD169/Siglec-1, Siglec-H, TMEM119, XCR1, CD24, CD103, CD64, CD83) and stained each brain individually with this antibody pool. \n\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\n\nif dict_datasets2consider['GSE148127 Mouse Brain'] == 'Yes':\n    key4dict = 'GSE148127 Mouse Brain'\n\n    if work_with_protein_data == 'Yes':\n        fn = '/kaggle/input/citeseqgse148127/GSE148127_ADT.counts.csv' # No CD53\n        df = pd.read_csv(fn,index_col = 0).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n    #     print('Original column names', df.columns)\n        def CDname_change(s):\n            r = s.split('-')[1]\n            if s == 'CITE-CCR2-CD192': r = 'CD192'\n            if s == 'CITE-PD-1': r = 'PD-1'\n            if s == 'CITE-PD-L1': r = 'PD-L1'\n            if s == 'CITE-I-A-I-E': r = 'I-A-I-E'\n            return r\n        l = [CDname_change(t) for t in df.columns]\n        df.columns = l\n        # print('Processed column names', df.columns)\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2))\n    \n    if work_with_rna_data == 'Yes':\n        fn = '/kaggle/input/citeseqgse148127/GSE148127_SCT.normalized.RNA.counts.csv'\n        df = pd.read_csv(fn, index_col = 0 ).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n\n    df = pd.DataFrame(index = df.index)\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect()    ","metadata":{"execution":{"iopub.status.busy":"2023-03-30T16:30:11.983500Z","iopub.execute_input":"2023-03-30T16:30:11.984061Z","iopub.status.idle":"2023-03-30T16:31:20.090433Z","shell.execute_reply.started":"2023-03-30T16:30:11.984019Z","shell.execute_reply":"2023-03-30T16:31:20.088917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE139891\n\nhttps://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE139891\n\n    # GSE139891 Single-cell transcriptomic analysis of human germinal center B cells\n    # Only 6 proteins are measured: ['CD184', 'CD44', 'CD69', 'CD83', 'CD9', 'CD196']\n    # 4513 cells GC3a\n    # 4777 cells:  GC3b \n\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc \nif dict_datasets2consider['GSE139891 HGC-Bcells'] == 'Yes':\n    for     key4dict in [ 'GSE139891 HGC-Bcells Part1', 'GSE139891 HGC-Bcells Part2']: \n    # GSE139891 Single-cell transcriptomic analysis of human germinal center B cells\n    # Only 6 proteins are measured: ['CD184', 'CD44', 'CD69', 'CD83', 'CD9', 'CD196']\n    # 4513 cells GC3a\n    # 4777 cells:  GC3b \n    \n        if work_with_protein_data == 'Yes':\n            if key4dict == 'GSE139891 HGC-Bcells Part1':\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560816_GC3a_Ab_counts.txt/GC3a_Ab_counts.txt' # Proteins (4513, 6)\n            else:\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560817_GC3b_Ab_counts.txt/GC3b_Ab_counts.txt' # proteins # 4777 cells:  GC3b\n\n            df = pd.read_csv(fn, index_col = 0 , sep = '\\t').T\n            print('Original shape:', df.shape)\n            df = df.iloc[:N_rows2take,:]\n            # print( df.info() )\n            l = [t.split('_')[1] for t in df.columns]\n            df.columns = l    \n            dict_df_prot[key4dict] = df\n            print('Protein data shape:', df.shape)\n            display(df.head(2)) \n\n        if work_with_rna_data == 'Yes':\n            if key4dict == 'GSE139891 HGC-Bcells Part1':\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560814_GC3a_umi.txt/counts_no_bad.txt' # RNA # (4513, 24019) \n            else:\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560815_GC3b_umi.txt/counts_no_bad.txt' # RNA 4777\n\n            df = pd.read_csv(fn, index_col = 0 , sep = '\\t').T\n            print('Original shape:', df.shape)\n            df = df.iloc[:N_rows2take,:]\n            l = [t.split(';')[1] for t in df.columns]\n            df.columns = l    \n            print('RNA data shape:', df.shape)\n            dict_df_rna[key4dict] = df\n            display(df.head(2)) \n\n\n        df = pd.DataFrame(index = df.index)\n        print('Meta data shape:', df.shape)\n        dict_df_meta[key4dict] = df\n        display(df.head(2)) \n\n        gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-24T18:59:39.609129Z","iopub.execute_input":"2023-03-24T18:59:39.609488Z","iopub.status.idle":"2023-03-24T19:00:32.153101Z","shell.execute_reply.started":"2023-03-24T18:59:39.609455Z","shell.execute_reply":"2023-03-24T19:00:32.151751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE160766 \n\n\tSingle-cell CITE-seq of two 9 month old BALBc mouse spleens\n    \nhttps://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE160766\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nif dict_datasets2consider['GSE160766 MouseSpleen'] == 'Yes':\n    # https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE160766\n    if work_with_protein_data == 'Yes':\n        key4dict = 'GSE160766 MouseSpleen'\n        fn = '/kaggle/input/cite-seqscrna-seq-proteins-gse160766/GSE160766_protein_expr_all.csv/GSE160766_protein_expr_all.csv'\n        df = pd.read_csv(fn).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        l = df.columns\n    #     print(l)\n        l2 = [t.split('_')[1] for t in l]\n        df.columns = l2\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2)) \n    \n    if work_with_rna_data == 'Yes':\n        fn = '/kaggle/input/cite-seqscrna-seq-proteins-gse160766/GSE160766_gene_matrix_all_new.csv/GSE160766_gene_matrix_all_new.csv'\n        df = pd.read_csv(fn).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n\n    df = pd.DataFrame(index = df.index)\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:00:32.154776Z","iopub.execute_input":"2023-03-24T19:00:32.155131Z","iopub.status.idle":"2023-03-24T19:00:45.405926Z","shell.execute_reply.started":"2023-03-24T19:00:32.155096Z","shell.execute_reply":"2023-03-24T19:00:45.404447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE155673 PBMC Covid","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nimport scipy.io\n\ntmp = dict_datasets2consider['GSE155673 PBMC Covid'] \nif (tmp == 'Yes') or (isinstance(tmp,(int,float, np.number) ) and tmp > 0) : #   'Yes':\n    N_data2load = 1 # \n    if (isinstance(tmp,(int,float, np.number) ) and tmp > 0) : \n        N_data2load = dict_datasets2consider['GSE155673 PBMC Covid']\n    \n    fn = '/kaggle/input/d/cosjo11/citeseqscrnaseqproteins-challenge-neurips2021/GSE155673_features.tsv/features.tsv'\n    df_genes_info = pd.read_csv(fn, sep = '\\t', header = None)\n    print(df_genes_info.shape)\n    display(df_genes_info.head(2))\n\n    m = df_genes_info[2] == 'Antibody Capture'\n    rna_names = list( df_genes_info[1][~m] )\n    rna_names[:2]\n    prot_names = list( df_genes_info[1][m] )\n    print(len(prot_names), 'Original prot names:', prot_names )\n    #print(prot_names)\n    def chg_name(s):\n        r = s.split('--')[0].split('_')[0]\n        if s.startswith('Isotype-control'): r = s\n        if s ==  'CD197-CCR7--G043H7-TSA' : r = 'CD197'\n        return r\n    prot_names = [chg_name(t) for t in prot_names]\n    print(len(prot_names), 'Transformed prot names:', prot_names )\n\n\n    list_data_ids = ['cov01', 'cov02', 'cov03', 'cov04', 'cov07', 'cov08', 'cov09', 'cov10', 'cov11', 'cov12', 'cov17', 'cov18']\n    for ii, str_id in enumerate( list_data_ids[:N_data2load] ):\n        print(ii, str_id)\n        key4dict = 'GSE155673 PBMC Covid ' + str_id\n\n        fn = '/kaggle/input/d/cosjo11/citeseqscrnaseqproteins-challenge-neurips2021/GSE155673_'+str_id+ '_matrix.mtx/matrix.mtx'\n        mtx = scipy.io.mmread(fn).T\n        print('Original shape rna+prots:', mtx.shape, type(mtx) )\n\n        fn = '/kaggle/input/d/cosjo11/citeseqscrnaseqproteins-challenge-neurips2021/GSE155673_'+str_id+'_barcodes.tsv/barcodes.tsv'\n        d = pd.read_csv(fn,header = None)\n        d\n\n        if work_with_protein_data == 'Yes':\n            df = pd.DataFrame(mtx.todense()[:N_rows2take,-len(prot_names ):], index = d[0].iloc[:N_rows2take] , columns = prot_names )\n            print(df.shape)\n            display(df.head(2))\n            dict_df_prot[key4dict] = df\n\n        if work_with_rna_data == 'Yes':\n            df = pd.DataFrame(mtx.todense()[:N_rows2take,:len(rna_names )], index = d[0].iloc[:N_rows2take] , columns = rna_names )\n            print(df.shape)\n            display(df.head(2))\n            dict_df_rna[key4dict] = df\n\n\n        df = pd.DataFrame(index = df.index)\n        print('Meta data shape:', df.shape)\n        dict_df_meta[key4dict] = df\n        display(df.head(2)) \n\n        gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:00:45.407325Z","iopub.execute_input":"2023-03-24T19:00:45.40858Z","iopub.status.idle":"2023-03-24T19:01:04.691243Z","shell.execute_reply.started":"2023-03-24T19:00:45.408526Z","shell.execute_reply":"2023-03-24T19:01:04.689792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis","metadata":{}},{"cell_type":"markdown","source":"## Set list of ligand repceptor pairs:","metadata":{}},{"cell_type":"code","source":"list_ligand_receptor_pairs = [ ('CD270', 'CD272'),('CD2', 'CD58'),('CD5', 'CD72'),('CD19', 'CD21'), ('CD28', 'CD80'), ('CD40', 'CD40L'), ('CD86', 'CD28'), ('CD95', 'CD95L'), ('CD154', 'CD40'), ('CD224', 'CD47'), ('CD319', 'CD319L') ]\n","metadata":{"execution":{"iopub.status.busy":"2023-03-30T17:10:43.891058Z","iopub.execute_input":"2023-03-30T17:10:43.892695Z","iopub.status.idle":"2023-03-30T17:10:43.908232Z","shell.execute_reply.started":"2023-03-30T17:10:43.892615Z","shell.execute_reply":"2023-03-30T17:10:43.906058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(list(dict_df_prot) )","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:01:04.700275Z","iopub.execute_input":"2023-03-24T19:01:04.700656Z","iopub.status.idle":"2023-03-24T19:01:04.715239Z","shell.execute_reply.started":"2023-03-24T19:01:04.70061Z","shell.execute_reply":"2023-03-24T19:01:04.713745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check pairs presence in different datasets","metadata":{}},{"cell_type":"code","source":"%%time\n\ndf_stat = pd.DataFrame(); IX = - 1\n\nfor ligand_receptor_pair in list_ligand_receptor_pairs:\n    print('ligand_receptor_pair:', ligand_receptor_pair )\n    for k in dict_df_meta:\n        df_prot = dict_df_prot[k]\n        flag0 = 0; \n        if ligand_receptor_pair[0] in df_prot.columns:            flag0 = 1;\n        flag1 = 0; \n        if ligand_receptor_pair[1] in df_prot.columns:            flag1 = 1;\n        if flag0 and flag1:\n            print(ligand_receptor_pair, 'found in dataset:', k)\n        df_stat.loc['Pair: '+str(ligand_receptor_pair ), k] = flag1 *flag0\n    \n    \ndf_stat    ","metadata":{"execution":{"iopub.status.busy":"2023-03-30T17:10:55.955835Z","iopub.execute_input":"2023-03-30T17:10:55.956272Z","iopub.status.idle":"2023-03-30T17:10:56.027584Z","shell.execute_reply.started":"2023-03-30T17:10:55.956235Z","shell.execute_reply":"2023-03-30T17:10:56.025980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Calculate correlations between ligand and repceptor for each pair","metadata":{}},{"cell_type":"code","source":"%%time\n\nfrom scipy import stats\n\ndf_stat = pd.DataFrame(); IX = - 1\n\nfor ligand_receptor_pair in list_ligand_receptor_pairs:\n    print('ligand_receptor_pair:', ligand_receptor_pair )\n    for k in dict_df_meta:\n        df_prot = dict_df_prot[k]\n        flag0 = 0; \n        if ligand_receptor_pair[0] in df_prot.columns:            flag0 = 1;\n        flag1 = 0; \n        if ligand_receptor_pair[1] in df_prot.columns:            flag1 = 1;\n        \n        if flag0 and flag1:\n            col0 = ligand_receptor_pair[0]\n            col1 = ligand_receptor_pair[1]\n            v0 = df_prot[col0]\n            v1 = df_prot[col1]\n            c = np.corrcoef(v0,v1)[0,1]\n            df_stat.loc['Pearson Correlation '+str(ligand_receptor_pair ), k] = c \n            res = stats.spearmanr(v0,v1)\n            df_stat.loc['Spearman Correlation '+str(ligand_receptor_pair ), k] = res[0]\n            df_stat.loc['Spearman Pvalue '+str(ligand_receptor_pair ), k] = res[1]\n    \n    \ndf_stat    ","metadata":{"execution":{"iopub.status.busy":"2023-03-30T17:15:04.161607Z","iopub.execute_input":"2023-03-30T17:15:04.162199Z","iopub.status.idle":"2023-03-30T17:15:04.338438Z","shell.execute_reply.started":"2023-03-30T17:15:04.162150Z","shell.execute_reply":"2023-03-30T17:15:04.336687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"execution":{"iopub.status.busy":"2023-03-24T19:01:04.777291Z","iopub.execute_input":"2023-03-24T19:01:04.777768Z","iopub.status.idle":"2023-03-24T19:01:04.785553Z","shell.execute_reply.started":"2023-03-24T19:01:04.77772Z","shell.execute_reply":"2023-03-24T19:01:04.78413Z"},"trusted":true},"execution_count":null,"outputs":[]}]}