{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nHere we look on common proteins for several datasets. \n\n\nWe choose HLA-DR (HLA-DRA) since it is present in the first three and not bad predicatable. We will analyse it later in more details.\n","metadata":{}},{"cell_type":"markdown","source":"# Key params","metadata":{}},{"cell_type":"code","source":"N_rows2take = int( 1e3 ) # Work only with first rows of the data - to speed up, avoid RAM crashes , etc... \n\nwork_with_rna_data = 'Yes'# If No - we will not store/work with protein data - only rna -  For some fast analysis we need only protein data, which much more smaller \nwork_with_protein_data = 'Yes' \n\n\ndict_datasets2consider = {}\ndict_datasets2consider['NIPS2022'] = 'Yes'\ndict_datasets2consider['NIPS2021'] = 'Yes'\ndict_datasets2consider['Kaggle2302'] = 'Yes'\ndict_datasets2consider['GSE148127 Mouse Brain'] = 'No'# 'No'\ndict_datasets2consider['GSE139891 HGC-Bcells'] = 'No'#'No'\ndict_datasets2consider['GSE160766 MouseSpleen'] = 'No'#'No'\ndict_datasets2consider['GSE155673 PBMC Covid'] = 0 # Can be a number - it indicates how many subdatsets from GSE155673 will be taked \n                                                   # There 12 subdatsets  \n                                                   # For case \"Yes\" we will take just one the first subdaset \n                                                   # subdatasets sizes about:  3000-9000 cells \n    \n\n# -----------------------------------------------------------------\n# Quick params modification : \n# -----------------------------------------------------------------\n\nif 0:  # Quick yes only for the first N\n    N = 3\n    for ii,k in enumerate(dict_datasets2consider):\n        if ii<N:\n            dict_datasets2consider[k] = 'Yes'\n        else:\n            dict_datasets2consider[k] = 'No'\n            \n\nif 0:  # Quick yes for all \n    for k in dict_datasets2consider:\n        dict_datasets2consider[k] = 'Yes'\n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:07.493927Z","iopub.execute_input":"2023-03-17T11:23:07.495506Z","iopub.status.idle":"2023-03-17T11:23:07.536860Z","shell.execute_reply.started":"2023-03-17T11:23:07.495431Z","shell.execute_reply":"2023-03-17T11:23:07.535535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparations","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport time \nt0start = time.time()\nimport gc\n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:07.539496Z","iopub.execute_input":"2023-03-17T11:23:07.540030Z","iopub.status.idle":"2023-03-17T11:23:08.986249Z","shell.execute_reply.started":"2023-03-17T11:23:07.539988Z","shell.execute_reply":"2023-03-17T11:23:08.985158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if 'research-project-01' not in os.path.join(dirname, filename): \n            print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-17T11:23:08.987779Z","iopub.execute_input":"2023-03-17T11:23:08.988529Z","iopub.status.idle":"2023-03-17T11:23:09.333944Z","shell.execute_reply.started":"2023-03-17T11:23:08.988475Z","shell.execute_reply":"2023-03-17T11:23:09.332717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scanpy\nimport scanpy as sc","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:09.336834Z","iopub.execute_input":"2023-03-17T11:23:09.337343Z","iopub.status.idle":"2023-03-17T11:23:31.099907Z","shell.execute_reply.started":"2023-03-17T11:23:09.337290Z","shell.execute_reply":"2023-03-17T11:23:31.098079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"dict_df_prot = {}\ndict_df_rna = {}\ndict_df_meta = {}\n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:31.101496Z","iopub.execute_input":"2023-03-17T11:23:31.101865Z","iopub.status.idle":"2023-03-17T11:23:31.107419Z","shell.execute_reply.started":"2023-03-17T11:23:31.101827Z","shell.execute_reply":"2023-03-17T11:23:31.106378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NIPS22-Kaggle ","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\n\nif dict_datasets2consider['NIPS2022'] == 'Yes':\n\n    key4dict = 'NIPS2022'\n    \n    DATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\n    if work_with_protein_data == 'Yes':\n        FP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\n        df = pd.read_hdf(FP_CITE_TRAIN_TARGETS, start=0, stop=N_rows2take, )\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2)) \n\n    # %%time\n    if work_with_rna_data == 'Yes':\n        filename_rna_data = '/kaggle/input/open-problems-multimodal/train_cite_inputs.h5'\n        df = pd.read_hdf(filename_rna_data, start=0, stop=N_rows2take,)\n        l = [t.split('_')[1] for t in df.columns]; df.columns = l \n        index_save = df.index\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n\n\n    fn = '/kaggle/input/open-problems-multimodal/metadata.csv'\n    df = pd.read_csv(fn, index_col = 0 );\n    mask = df.index.isin(index_save); df = df[mask]\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect() \n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:31.109027Z","iopub.execute_input":"2023-03-17T11:23:31.110478Z","iopub.status.idle":"2023-03-17T11:23:33.125267Z","shell.execute_reply.started":"2023-03-17T11:23:31.110409Z","shell.execute_reply":"2023-03-17T11:23:33.123668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NIPS21","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\n\nif dict_datasets2consider['NIPS2021'] == 'Yes':\n    key4dict = 'NIPS2021'\n    fn = '/kaggle/input/citeseqscrnaseqproteins-challenge-neurips2021/GSE194122_openproblems_neurips2021_cite_BMMC_processed.h5ad'\n    adata = sc.read(fn)\n    adata.var_names_make_unique()\n    print(adata)\n    \n    if work_with_protein_data == 'Yes':\n        mask = adata.var['feature_types' ] == 'ADT'\n        X = adata[:N_rows2take,mask].X.todense()\n        df = pd.DataFrame(X, columns = list(adata.var['feature_types' ][mask].index)  )\n        l = [t.replace('-1','') for t in df.columns]\n        df.columns = l\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        print( 'Protein data shape:', df.shape)\n        dict_df_prot[key4dict] = df\n        display(df.head(2)) #df\n    \n    if work_with_rna_data == 'Yes':\n        mask = adata.var['feature_types' ] != 'ADT'\n        X = adata[:N_rows2take,mask].X.todense()\n        df = pd.DataFrame(X, columns = list(adata.var['feature_types' ][mask].index)  )\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n    \n    df = adata.obs.iloc[:N_rows2take,:]\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect()\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:33.127329Z","iopub.execute_input":"2023-03-17T11:23:33.128798Z","iopub.status.idle":"2023-03-17T11:23:53.215090Z","shell.execute_reply.started":"2023-03-17T11:23:33.128742Z","shell.execute_reply":"2023-03-17T11:23:53.213402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kaggle 2023 02 community competition \n\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nif dict_datasets2consider['Kaggle2302'] == 'Yes':\n    key4dict = 'Kaggle2302'\n\n    if work_with_protein_data == 'Yes':\n        df = pd.read_csv('/kaggle/input/machine-learning-challenge-2-prediction/training_set_adt.csv', index_col = 0).T # Train targets     \n        print('Original shape', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2)) \n\n    if work_with_rna_data == 'Yes':\n        df = pd.read_csv('/kaggle/input/machine-learning-challenge-2-prediction/training_set_rna.csv', index_col = 0).T    # Train features \n        print('Original shape', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df.iloc[:N_rows2take,:]\n        display(df.head(2)) \n\n\n    df = pd.DataFrame(index = df.index)\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df.iloc[:N_rows2take,:]\n    display(df.head(2)) \n    gc.collect()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:53.217472Z","iopub.execute_input":"2023-03-17T11:23:53.217927Z","iopub.status.idle":"2023-03-17T11:23:54.794077Z","shell.execute_reply.started":"2023-03-17T11:23:53.217883Z","shell.execute_reply":"2023-03-17T11:23:54.792263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE148127 mouse \n\nhttps://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE148127\n\nWe used CITE-seq (10x Gemoics-based) to profile and compare the transcriptomes and cell surface expression of a set of immune markers of naïve brain myeloid cells from young-adult (12 weeks old) wild type C57Bl/6 mice, aged (72 weeks old) wild type C57Bl/6 mice and young-adult and aged C57Bl/6 mice with antibiotics induced gut microbiota depletion. In one sequencing round, we sequenced a total of 12 different mouse brains (3 young control, 3 aged control, 3 young depleted gut microbiota (ABX) and 3 aged depleted gut microbiota (ABX)).\n\n\nWe created an antibody pool consisting of 31 different antibodies (CCR2/CD192, C117/c-kit, CD11b, CD11c, CD172a/SIRPα, CD25, CD3, CD38, CD4, CD44, CD45, CD45R/B220, CD86, CD8a, CD90.1, Cx3cr1, F4/80, I-A/I-E, Ly6C, Ly6G, NK1.1, PD-1, PD-L1, CD169/Siglec-1, Siglec-H, TMEM119, XCR1, CD24, CD103, CD64, CD83) and stained each brain individually with this antibody pool. \n\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\n\nif dict_datasets2consider['GSE148127 Mouse Brain'] == 'Yes':\n    key4dict = 'GSE148127 Mouse Brain'\n\n    if work_with_protein_data == 'Yes':\n        fn = '/kaggle/input/citeseqgse148127/GSE148127_ADT.counts.csv' # No CD53\n        df = pd.read_csv(fn,index_col = 0).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n    #     print('Original column names', df.columns)\n        def CDname_change(s):\n            r = s.split('-')[1]\n            if s == 'CITE-CCR2-CD192': r = 'CD192'\n            if s == 'CITE-PD-1': r = 'PD-1'\n            if s == 'CITE-PD-L1': r = 'PD-L1'\n            if s == 'CITE-I-A-I-E': r = 'I-A-I-E'\n            return r\n        l = [CDname_change(t) for t in df.columns]\n        df.columns = l\n        # print('Processed column names', df.columns)\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2))\n    \n    if work_with_rna_data == 'Yes':\n        fn = '/kaggle/input/citeseqgse148127/GSE148127_SCT.normalized.RNA.counts.csv'\n        df = pd.read_csv(fn, index_col = 0 ).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n\n    df = pd.DataFrame(index = df.index)\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect()    ","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:54.796014Z","iopub.execute_input":"2023-03-17T11:23:54.796558Z","iopub.status.idle":"2023-03-17T11:23:54.811076Z","shell.execute_reply.started":"2023-03-17T11:23:54.796500Z","shell.execute_reply":"2023-03-17T11:23:54.809023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE139891\n\nhttps://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE139891\n\n    # GSE139891 Single-cell transcriptomic analysis of human germinal center B cells\n    # Only 6 proteins are measured: ['CD184', 'CD44', 'CD69', 'CD83', 'CD9', 'CD196']\n    # 4513 cells GC3a\n    # 4777 cells:  GC3b \n\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc \nif dict_datasets2consider['GSE139891 HGC-Bcells'] == 'Yes':\n    for     key4dict in [ 'GSE139891 HGC-Bcells Part1', 'GSE139891 HGC-Bcells Part2']: \n    # GSE139891 Single-cell transcriptomic analysis of human germinal center B cells\n    # Only 6 proteins are measured: ['CD184', 'CD44', 'CD69', 'CD83', 'CD9', 'CD196']\n    # 4513 cells GC3a\n    # 4777 cells:  GC3b \n    \n        if work_with_protein_data == 'Yes':\n            if key4dict == 'GSE139891 HGC-Bcells Part1':\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560816_GC3a_Ab_counts.txt/GC3a_Ab_counts.txt' # Proteins (4513, 6)\n            else:\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560817_GC3b_Ab_counts.txt/GC3b_Ab_counts.txt' # proteins # 4777 cells:  GC3b\n\n            df = pd.read_csv(fn, index_col = 0 , sep = '\\t').T\n            print('Original shape:', df.shape)\n            df = df.iloc[:N_rows2take,:]\n            # print( df.info() )\n            l = [t.split('_')[1] for t in df.columns]\n            df.columns = l    \n            dict_df_prot[key4dict] = df\n            print('Protein data shape:', df.shape)\n            display(df.head(2)) \n\n        if work_with_rna_data == 'Yes':\n            if key4dict == 'GSE139891 HGC-Bcells Part1':\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560814_GC3a_umi.txt/counts_no_bad.txt' # RNA # (4513, 24019) \n            else:\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560815_GC3b_umi.txt/counts_no_bad.txt' # RNA 4777\n\n            df = pd.read_csv(fn, index_col = 0 , sep = '\\t').T\n            print('Original shape:', df.shape)\n            df = df.iloc[:N_rows2take,:]\n            l = [t.split(';')[1] for t in df.columns]\n            df.columns = l    \n            print('RNA data shape:', df.shape)\n            dict_df_rna[key4dict] = df\n            display(df.head(2)) \n\n\n        df = pd.DataFrame(index = df.index)\n        print('Meta data shape:', df.shape)\n        dict_df_meta[key4dict] = df\n        display(df.head(2)) \n\n        gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:54.814879Z","iopub.execute_input":"2023-03-17T11:23:54.815628Z","iopub.status.idle":"2023-03-17T11:23:54.833220Z","shell.execute_reply.started":"2023-03-17T11:23:54.815573Z","shell.execute_reply":"2023-03-17T11:23:54.832108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE160766 \n\n\tSingle-cell CITE-seq of two 9 month old BALBc mouse spleens\n    \nhttps://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE160766\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nif dict_datasets2consider['GSE160766 MouseSpleen'] == 'Yes':\n    # https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE160766\n    if work_with_protein_data == 'Yes':\n        key4dict = 'GSE160766 MouseSpleen'\n        fn = '/kaggle/input/cite-seqscrna-seq-proteins-gse160766/GSE160766_protein_expr_all.csv/GSE160766_protein_expr_all.csv'\n        df = pd.read_csv(fn).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        l = df.columns\n    #     print(l)\n        l2 = [t.split('_')[1] for t in l]\n        df.columns = l2\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2)) \n    \n    if work_with_rna_data == 'Yes':\n        fn = '/kaggle/input/cite-seqscrna-seq-proteins-gse160766/GSE160766_gene_matrix_all_new.csv/GSE160766_gene_matrix_all_new.csv'\n        df = pd.read_csv(fn).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n\n    df = pd.DataFrame(index = df.index)\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:54.834816Z","iopub.execute_input":"2023-03-17T11:23:54.835793Z","iopub.status.idle":"2023-03-17T11:23:54.856963Z","shell.execute_reply.started":"2023-03-17T11:23:54.835745Z","shell.execute_reply":"2023-03-17T11:23:54.855268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE155673 PBMC Covid","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nimport scipy.io\n\ntmp = dict_datasets2consider['GSE155673 PBMC Covid'] \nif (tmp == 'Yes') or (isinstance(tmp,(int,float, np.number) ) and tmp > 0) : #   'Yes':\n    N_data2load = 1 # \n    if (isinstance(tmp,(int,float, np.number) ) and tmp > 0) : \n        N_data2load = dict_datasets2consider['GSE155673 PBMC Covid']\n    \n    fn = '/kaggle/input/d/cosjo11/citeseqscrnaseqproteins-challenge-neurips2021/GSE155673_features.tsv/features.tsv'\n    df_genes_info = pd.read_csv(fn, sep = '\\t', header = None)\n    print(df_genes_info.shape)\n    display(df_genes_info.head(2))\n\n    m = df_genes_info[2] == 'Antibody Capture'\n    rna_names = list( df_genes_info[1][~m] )\n    rna_names[:2]\n    prot_names = list( df_genes_info[1][m] )\n    print(len(prot_names), 'Original prot names:', prot_names )\n    #print(prot_names)\n    def chg_name(s):\n        r = s.split('--')[0].split('_')[0]\n        if s.startswith('Isotype-control'): r = s\n        if s ==  'CD197-CCR7--G043H7-TSA' : r = 'CD197'\n        return r\n    prot_names = [chg_name(t) for t in prot_names]\n    print(len(prot_names), 'Transformed prot names:', prot_names )\n\n\n    list_data_ids = ['cov01', 'cov02', 'cov03', 'cov04', 'cov07', 'cov08', 'cov09', 'cov10', 'cov11', 'cov12', 'cov17', 'cov18']\n    for ii, str_id in enumerate( list_data_ids[:N_data2load] ):\n        print(ii, str_id)\n        key4dict = 'GSE155673 PBMC Covid ' + str_id\n\n        fn = '/kaggle/input/d/cosjo11/citeseqscrnaseqproteins-challenge-neurips2021/GSE155673_'+str_id+ '_matrix.mtx/matrix.mtx'\n        mtx = scipy.io.mmread(fn).T\n        print('Original shape rna+prots:', mtx.shape, type(mtx) )\n\n        fn = '/kaggle/input/d/cosjo11/citeseqscrnaseqproteins-challenge-neurips2021/GSE155673_'+str_id+'_barcodes.tsv/barcodes.tsv'\n        d = pd.read_csv(fn,header = None)\n        d\n\n        if work_with_protein_data == 'Yes':\n            df = pd.DataFrame(mtx.todense()[:N_rows2take,-len(prot_names ):], index = d[0].iloc[:N_rows2take] , columns = prot_names )\n            print(df.shape)\n            display(df.head(2))\n            dict_df_prot[key4dict] = df\n\n        if work_with_rna_data == 'Yes':\n            df = pd.DataFrame(mtx.todense()[:N_rows2take,:len(rna_names )], index = d[0].iloc[:N_rows2take] , columns = rna_names )\n            print(df.shape)\n            display(df.head(2))\n            dict_df_rna[key4dict] = df\n\n\n        df = pd.DataFrame(index = df.index)\n        print('Meta data shape:', df.shape)\n        dict_df_meta[key4dict] = df\n        display(df.head(2)) \n\n        gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:54.859294Z","iopub.execute_input":"2023-03-17T11:23:54.859774Z","iopub.status.idle":"2023-03-17T11:23:54.923894Z","shell.execute_reply.started":"2023-03-17T11:23:54.859723Z","shell.execute_reply":"2023-03-17T11:23:54.922496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis","metadata":{}},{"cell_type":"code","source":"print(list(dict_df_prot) )","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:54.925622Z","iopub.execute_input":"2023-03-17T11:23:54.926321Z","iopub.status.idle":"2023-03-17T11:23:54.932733Z","shell.execute_reply.started":"2023-03-17T11:23:54.926274Z","shell.execute_reply":"2023-03-17T11:23:54.931203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ndf_stat = pd.DataFrame(); IX = - 1\n\nfor k in dict_df_meta:\n    IX += 1\n    df_stat.loc[IX,'Data'] = str(k)\n    if k in dict_df_prot.keys():\n        df = dict_df_prot[k]\n        df_stat.loc[IX,'Protein Data Shape'] = str(df.shape)\n        df_stat.loc[IX,'Protein Raw Counts'] = np.all( (df.sum()).astype(int) ==  df.sum() )\n        df_stat.loc[IX,'Protein %Zeros'] = np.round( (df == 0).sum().sum()/df.shape[0]/df.shape[1]*100, 1)\n    if k in dict_df_rna.keys():\n        df = dict_df_rna[k]\n        df_stat.loc[IX,'RNA Data Shape'] = str(df.shape)\n        df_stat.loc[IX,'RNA Raw Counts'] = np.all( (df.sum()).astype(int) ==  df.sum() )\n        df_stat.loc[IX,'RNA %Zeros'] = np.round( (df == 0).sum().sum()/df.shape[0]/df.shape[1]*100, 1)\n    df = dict_df_meta[k]\n    df_stat.loc[IX,'Meta Data Shape'] = str(df.shape)\ndf_stat    ","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:54.934638Z","iopub.execute_input":"2023-03-17T11:23:54.935503Z","iopub.status.idle":"2023-03-17T11:23:55.521910Z","shell.execute_reply.started":"2023-03-17T11:23:54.935446Z","shell.execute_reply":"2023-03-17T11:23:55.520348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_CD_proteinName_to_CD_rnaName_Kaggle2302 = { \n'CD3': 'CD3D',\n'CD127-IL7Ra' : 'IL7R',\n'CD14' : 'CD14',\n'CD161' : 'KLRB1',\n'CD27' : 'CD27',\n'CD8a' : 'CD8A',\n'CD79b' : 'CD79B',\n'CD16' : 'FCGR3A',\n'CD45RO' : 'PTPRC',\n'CD197-CCR7' : 'CCR7',\n'CD45RA' : 'PTPRC',\n'HLA.DR' : 'HLA-DRA'}\nprint(len(dict_CD_proteinName_to_CD_rnaName_Kaggle2302), 'proteins out of 25 have corresponding \"own\" rna in the rna dataframe') ","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:51:16.432901Z","iopub.execute_input":"2023-03-17T11:51:16.433418Z","iopub.status.idle":"2023-03-17T11:51:16.442135Z","shell.execute_reply.started":"2023-03-17T11:51:16.433373Z","shell.execute_reply":"2023-03-17T11:51:16.440556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Top predicatable for Kaggle-NIPS2022 dataset : \n\n    Name\tr2score\n    1\tCD41\t0.823414\n    2\tCD36\t0.738582\n    3\tCD32\t0.72686\n    4\tCD88\t0.721497\n    5\tCD71\t0.719141\n    6\tCD48\t0.711979\n    7\tCD62L\t0.684082\n    8\tCD45RA\t0.639678\n    9\tCD49b\t0.627032\n    10\tCD82\t0.582508\n    11\tCD38\t0.580875\n    12\tCD11a\t0.570223\n    13\tCD115\t0.557473\n    14\tCD244\t0.547887\n    15\tCD31\t0.507543\n    16\tCD44\t0.50609\n    17\tFceRIa\t0.492126\n    18\tCD13\t0.484275\n    19\tCD112\t0.461958\n    20\tHLA-DR\t0.455588\n\nFrom here(?): https://www.kaggle.com/code/alexandervc/mmscel-blend-analysis?scriptVersionId=112307232&cellId=55\n","metadata":{}},{"cell_type":"markdown","source":"# Look on common CD for various datasets","metadata":{}},{"cell_type":"code","source":"%%time\n\nc = 0;\nfor k in dict_df_meta:\n    c += 1 \n    df = dict_df_prot[k]\n    if c == 1:\n        s = set(df.columns)\n    else:\n        s = s & set(df.columns)\nprint('Common CD proteins for all datasets: ', len(s), s) \nprint()\ns_save = s \n\ns = s # + [dict_CD_proteinName_to_CD_rnaName_Kaggle2302(t) for t in s ]\nfor k in dict_df_meta:\n    c += 1 \n    df = dict_df_rna[k]\n    s = s & set(df.columns)\nprint(len(s), s) \nprint('Common which are also have RNA in : ', len(s), s) \nprint()\n\ns = s_save\ns = set( list(s) + [dict_CD_proteinName_to_CD_rnaName_Kaggle2302[t] for t in s if t in dict_CD_proteinName_to_CD_rnaName_Kaggle2302.keys() ] )\nfor k in dict_df_meta:\n    c += 1 \n    df = dict_df_rna[k]\n    s = s & set(df.columns)\nprint(len(s), s) \nprint('Common which are also have RNA in : ', len(s), s) \nprint()\n\n\nfor k in dict_df_meta:\n    c += 1 \n    df = dict_df_rna[k]\n    t = ('KLRB1' in df.columns ) \n    print(t, k)# s = s & set(df.columns)\n\nfor k in dict_df_meta:\n    c += 1 \n    df = dict_df_rna[k]\n    t = ('KLRB1' in df.columns ) \n    print(t, k)# s = s & set(df.columns)\n    \nprint()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:53:36.677618Z","iopub.execute_input":"2023-03-17T11:53:36.679191Z","iopub.status.idle":"2023-03-17T11:53:36.709752Z","shell.execute_reply.started":"2023-03-17T11:53:36.679091Z","shell.execute_reply":"2023-03-17T11:53:36.708072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"CD14 is quite well predictable for Kaglle2302 dataset - correlation 0.97(!) :  https://www.kaggle.com/code/alexandervc/cite-seq2023-eda?scriptVersionId=122454182&cellId=65 \n","metadata":{}},{"cell_type":"markdown","source":"# Found: HLA-DR(A) present everywhere is not bad predictable on NIPS22 (0.46 ) and on Kaggle2302 ","metadata":{}},{"cell_type":"code","source":"for k in dict_df_meta:\n    df = dict_df_rna[k]\n    t = ('HLA.DR' in df.columns ) or ('HLA-DR' in df.columns ) or ('HLA-DRA' in df.columns )\n    print(t, k)# s = s & set(df.columns)\n    df = dict_df_prot[k]\n    t = ('HLA.DR' in df.columns ) or ('HLA-DR' in df.columns )  or ('HLA-DRA' in df.columns ) \n    print(t, k)# s = s & set(df.columns)\n    print()","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:01:20.111887Z","iopub.execute_input":"2023-03-17T12:01:20.113499Z","iopub.status.idle":"2023-03-17T12:01:20.123226Z","shell.execute_reply.started":"2023-03-17T12:01:20.113426Z","shell.execute_reply":"2023-03-17T12:01:20.121652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ncol1 = 'HLA.DR'\ncol2 = 'HLA-DRA'\nc = 0;\nfor k in dict_df_meta:\n    c += 1 \n    if (col1 not in  dict_df_prot[k].columns ) or ( col2 not in  dict_df_rna[k].columns ): continue \n    df = dict_df_prot[k]\n    x = df[col1]\n    df = dict_df_rna[k]\n    y = df[col2]\n    sns.scatterplot(x=x, y = y)\n    plt.title(k)\n    plt.show()\n\ncol1 = 'HLA-DR'\ncol2 = 'HLA-DRA'\nc = 0;\nfor k in dict_df_meta:\n    c += 1 \n    if (col1 not in  dict_df_prot[k].columns ) or ( col2 not in  dict_df_rna[k].columns ): continue \n    df = dict_df_prot[k]\n    x = df[col1]\n    df = dict_df_rna[k]\n    y = df[col2]\n    sns.scatterplot(x=x, y = y)\n    plt.title(k)\n    plt.show()    ","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:06:22.562081Z","iopub.execute_input":"2023-03-17T12:06:22.563702Z","iopub.status.idle":"2023-03-17T12:06:23.166359Z","shell.execute_reply.started":"2023-03-17T12:06:22.563644Z","shell.execute_reply":"2023-03-17T12:06:23.164914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Some RNA from common list are completely zero on some datasets .","metadata":{}},{"cell_type":"code","source":"%%time\n\ncol = 'CD27'\nc = 0;\nfor k in dict_df_meta:\n    c += 1 \n    df = dict_df_prot[k]\n    x = df[col]\n    df = dict_df_rna[k]\n    y = df[col]\n    sns.scatterplot(x=x, y = y)\n    plt.title(k)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:33:58.395060Z","iopub.execute_input":"2023-03-17T11:33:58.396707Z","iopub.status.idle":"2023-03-17T11:33:59.108859Z","shell.execute_reply.started":"2023-03-17T11:33:58.396635Z","shell.execute_reply":"2023-03-17T11:33:59.107461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ncol = 'CD14'\nc = 0;\nfor k in dict_df_meta:\n    c += 1 \n    df = dict_df_prot[k]\n    x = df[col]\n    df = dict_df_rna[k]\n    y = df[col]\n    sns.scatterplot(x=x, y = y)\n    plt.title(k)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:33:32.235626Z","iopub.execute_input":"2023-03-17T11:33:32.237237Z","iopub.status.idle":"2023-03-17T11:33:32.967184Z","shell.execute_reply.started":"2023-03-17T11:33:32.237164Z","shell.execute_reply":"2023-03-17T11:33:32.965646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Look on CD45","metadata":{}},{"cell_type":"code","source":"%%time\n\ndf_stat = pd.DataFrame(); IX = -1\nfor col1 in ['CD45RA', 'CD45RO', 'CD45']:\n    col2 = 'PTPRC'\n    c = 0;\n    for k in dict_df_meta:\n        c += 1 \n        if (col1 not in  dict_df_prot[k].columns ) or ( col2 not in  dict_df_rna[k].columns ): continue \n        df = dict_df_prot[k]\n        x = df[col1]\n        df = dict_df_rna[k]\n        y = df[col2]\n        sns.scatterplot(x=x, y = y)\n        plt.title(k)\n        plt.show()\n        \n        IX+=1\n        df_stat.loc[IX,'Data'] = k\n        df_stat.loc[IX,'Prot'] = col1\n        df_stat.loc[IX,'RNA'] = col2\n        df_stat.loc[IX,'Corr'] = np.corrcoef(x,y)[0,1] \n        \n        \ndf_stat        \n","metadata":{"execution":{"iopub.status.busy":"2023-03-17T12:29:27.482627Z","iopub.execute_input":"2023-03-17T12:29:27.483272Z","iopub.status.idle":"2023-03-17T12:29:29.565289Z","shell.execute_reply.started":"2023-03-17T12:29:27.483215Z","shell.execute_reply":"2023-03-17T12:29:29.562433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"execution":{"iopub.status.busy":"2023-03-17T11:23:55.523918Z","iopub.execute_input":"2023-03-17T11:23:55.524584Z","iopub.status.idle":"2023-03-17T11:23:55.533190Z","shell.execute_reply.started":"2023-03-17T11:23:55.524518Z","shell.execute_reply":"2023-03-17T11:23:55.531373Z"},"trusted":true},"execution_count":null,"outputs":[]}]}