{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\n\nLoad both RNA and protein part for several CITE-seq datasets.\n\nThere is basically no analysis - notebook planned to be template for futher notebooks with analysis\n\nData is stored in dict_df with a key - which contain info on a dataset.\n\nTo load  data for the first 6 datasets takes about 236 seconds. With restriction to 10 000 samples - 156 seconds.\n\nPS\n\nRelated script to analyse only protein part of the data: https://www.kaggle.com/code/alexandervc/many-citeseq-analyse-protein-data-01\n","metadata":{}},{"cell_type":"markdown","source":"# Key params","metadata":{}},{"cell_type":"code","source":"N_rows2take = int( 1e3 ) # Work only with first rows of the data - to speed up, avoid RAM crashes , etc... \n\nwork_with_rna_data = 'Yes'# If No - we will not store/work with protein data - only rna -  For some fast analysis we need only protein data, which much more smaller \nwork_with_protein_data = 'Yes' \n\n\ndict_datasets2consider = {}\ndict_datasets2consider['NIPS2022'] = 'Yes'\ndict_datasets2consider['NIPS2021'] = 'Yes'\ndict_datasets2consider['Kaggle2302'] = 'Yes'\ndict_datasets2consider['GSE148127 Mouse Brain'] = 'Yes'# 'No'\ndict_datasets2consider['GSE139891 HGC-Bcells'] = 'Yes'#'No'\ndict_datasets2consider['GSE160766 MouseSpleen'] = 'Yes'#'No'\ndict_datasets2consider['GSE155673 PBMC Covid'] = 3 # Can be a number - it indicates how many subdatsets from GSE155673 will be taked \n                                                   # There 12 subdatsets  \n                                                   # For case \"Yes\" we will take just one the first subdaset \n                                                   # subdatasets sizes about:  3000-9000 cells \n    \n\n# -----------------------------------------------------------------\n# Quick params modification : \n# -----------------------------------------------------------------\n\nif 0:  # Quick yes only for the first N\n    N = 3\n    for ii,k in enumerate(dict_datasets2consider):\n        if ii<N:\n            dict_datasets2consider[k] = 'Yes'\n        else:\n            dict_datasets2consider[k] = 'No'\n            \n\nif 0:  # Quick yes for all \n    for k in dict_datasets2consider:\n        dict_datasets2consider[k] = 'Yes'\n","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:43:15.490677Z","iopub.execute_input":"2023-03-14T21:43:15.491523Z","iopub.status.idle":"2023-03-14T21:43:15.533533Z","shell.execute_reply.started":"2023-03-14T21:43:15.491470Z","shell.execute_reply":"2023-03-14T21:43:15.532442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparations","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport time \nt0start = time.time()\nimport gc\n","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:43:15.535526Z","iopub.execute_input":"2023-03-14T21:43:15.536219Z","iopub.status.idle":"2023-03-14T21:43:17.013633Z","shell.execute_reply.started":"2023-03-14T21:43:15.536179Z","shell.execute_reply":"2023-03-14T21:43:17.012198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if 'research-project-01' not in os.path.join(dirname, filename): \n            print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-14T21:43:17.016592Z","iopub.execute_input":"2023-03-14T21:43:17.017150Z","iopub.status.idle":"2023-03-14T21:43:17.328609Z","shell.execute_reply.started":"2023-03-14T21:43:17.017096Z","shell.execute_reply":"2023-03-14T21:43:17.327113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scanpy\nimport scanpy as sc","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:43:17.332476Z","iopub.execute_input":"2023-03-14T21:43:17.333005Z","iopub.status.idle":"2023-03-14T21:43:38.054567Z","shell.execute_reply.started":"2023-03-14T21:43:17.332955Z","shell.execute_reply":"2023-03-14T21:43:38.052878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"dict_df_prot = {}\ndict_df_rna = {}\ndict_df_meta = {}\n","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:43:38.056984Z","iopub.execute_input":"2023-03-14T21:43:38.058233Z","iopub.status.idle":"2023-03-14T21:43:38.064773Z","shell.execute_reply.started":"2023-03-14T21:43:38.058184Z","shell.execute_reply":"2023-03-14T21:43:38.062779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NIPS22-Kaggle ","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\n\nif dict_datasets2consider['NIPS2022'] == 'Yes':\n\n    key4dict = 'NIPS2022'\n    \n    DATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\n    if work_with_protein_data == 'Yes':\n        FP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\n        df = pd.read_hdf(FP_CITE_TRAIN_TARGETS, start=0, stop=N_rows2take, )\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2)) \n\n    # %%time\n    if work_with_rna_data == 'Yes':\n        filename_rna_data = '/kaggle/input/open-problems-multimodal/train_cite_inputs.h5'\n        df = pd.read_hdf(filename_rna_data, start=0, stop=N_rows2take,)\n        l = [t.split('_')[1] for t in df.columns]; df.columns = l \n        index_save = df.index\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n\n\n    fn = '/kaggle/input/open-problems-multimodal/metadata.csv'\n    df = pd.read_csv(fn, index_col = 0 );\n    mask = df.index.isin(index_save); df = df[mask]\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect() \n","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:43:38.067103Z","iopub.execute_input":"2023-03-14T21:43:38.067487Z","iopub.status.idle":"2023-03-14T21:43:40.048399Z","shell.execute_reply.started":"2023-03-14T21:43:38.067453Z","shell.execute_reply":"2023-03-14T21:43:40.046496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NIPS21","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\n\nif dict_datasets2consider['NIPS2021'] == 'Yes':\n    key4dict = 'NIPS2021'\n    fn = '/kaggle/input/citeseqscrnaseqproteins-challenge-neurips2021/GSE194122_openproblems_neurips2021_cite_BMMC_processed.h5ad'\n    adata = sc.read(fn)\n    adata.var_names_make_unique()\n    print(adata)\n    \n    if work_with_protein_data == 'Yes':\n        mask = adata.var['feature_types' ] == 'ADT'\n        X = adata[:N_rows2take,mask].X.todense()\n        df = pd.DataFrame(X, columns = list(adata.var['feature_types' ][mask].index)  )\n        l = [t.replace('-1','') for t in df.columns]\n        df.columns = l\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        print( 'Protein data shape:', df.shape)\n        dict_df_prot[key4dict] = df\n        display(df.head(2)) #df\n    \n    if work_with_rna_data == 'Yes':\n        mask = adata.var['feature_types' ] != 'ADT'\n        X = adata[:N_rows2take,mask].X.todense()\n        df = pd.DataFrame(X, columns = list(adata.var['feature_types' ][mask].index)  )\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n    \n    df = adata.obs.iloc[:N_rows2take,:]\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect()\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:43:40.050644Z","iopub.execute_input":"2023-03-14T21:43:40.051066Z","iopub.status.idle":"2023-03-14T21:43:57.603515Z","shell.execute_reply.started":"2023-03-14T21:43:40.051030Z","shell.execute_reply":"2023-03-14T21:43:57.602059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kaggle 2023 02 community competition \n\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nif dict_datasets2consider['Kaggle2302'] == 'Yes':\n    key4dict = 'Kaggle2302'\n\n    if work_with_protein_data == 'Yes':\n        df = pd.read_csv('/kaggle/input/machine-learning-challenge-2-prediction/training_set_adt.csv', index_col = 0).T # Train targets     \n        print('Original shape', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2)) \n\n    if work_with_rna_data == 'Yes':\n        df = pd.read_csv('/kaggle/input/machine-learning-challenge-2-prediction/training_set_rna.csv', index_col = 0).T    # Train features \n        print('Original shape', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df.iloc[:N_rows2take,:]\n        display(df.head(2)) \n\n\n    df = pd.DataFrame(index = df.index)\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df.iloc[:N_rows2take,:]\n    display(df.head(2)) \n    gc.collect()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:43:57.605741Z","iopub.execute_input":"2023-03-14T21:43:57.606153Z","iopub.status.idle":"2023-03-14T21:43:59.167284Z","shell.execute_reply.started":"2023-03-14T21:43:57.606117Z","shell.execute_reply":"2023-03-14T21:43:59.165946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE148127 mouse \n\nhttps://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE148127\n\nWe used CITE-seq (10x Gemoics-based) to profile and compare the transcriptomes and cell surface expression of a set of immune markers of naïve brain myeloid cells from young-adult (12 weeks old) wild type C57Bl/6 mice, aged (72 weeks old) wild type C57Bl/6 mice and young-adult and aged C57Bl/6 mice with antibiotics induced gut microbiota depletion. In one sequencing round, we sequenced a total of 12 different mouse brains (3 young control, 3 aged control, 3 young depleted gut microbiota (ABX) and 3 aged depleted gut microbiota (ABX)).\n\n\nWe created an antibody pool consisting of 31 different antibodies (CCR2/CD192, C117/c-kit, CD11b, CD11c, CD172a/SIRPα, CD25, CD3, CD38, CD4, CD44, CD45, CD45R/B220, CD86, CD8a, CD90.1, Cx3cr1, F4/80, I-A/I-E, Ly6C, Ly6G, NK1.1, PD-1, PD-L1, CD169/Siglec-1, Siglec-H, TMEM119, XCR1, CD24, CD103, CD64, CD83) and stained each brain individually with this antibody pool. \n\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\n\nif dict_datasets2consider['GSE148127 Mouse Brain'] == 'Yes':\n    key4dict = 'GSE148127 Mouse Brain'\n\n    if work_with_protein_data == 'Yes':\n        fn = '/kaggle/input/citeseqgse148127/GSE148127_ADT.counts.csv' # No CD53\n        df = pd.read_csv(fn,index_col = 0).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n    #     print('Original column names', df.columns)\n        def CDname_change(s):\n            r = s.split('-')[1]\n            if s == 'CITE-CCR2-CD192': r = 'CD192'\n            if s == 'CITE-PD-1': r = 'PD-1'\n            if s == 'CITE-PD-L1': r = 'PD-L1'\n            if s == 'CITE-I-A-I-E': r = 'I-A-I-E'\n            return r\n        l = [CDname_change(t) for t in df.columns]\n        df.columns = l\n        # print('Processed column names', df.columns)\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2))\n    \n    if work_with_rna_data == 'Yes':\n        fn = '/kaggle/input/citeseqgse148127/GSE148127_SCT.normalized.RNA.counts.csv'\n        df = pd.read_csv(fn, index_col = 0 ).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n\n    df = pd.DataFrame(index = df.index)\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect()    ","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:43:59.168753Z","iopub.execute_input":"2023-03-14T21:43:59.169107Z","iopub.status.idle":"2023-03-14T21:45:07.638879Z","shell.execute_reply.started":"2023-03-14T21:43:59.169061Z","shell.execute_reply":"2023-03-14T21:45:07.637453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE139891\n\nhttps://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE139891\n\n    # GSE139891 Single-cell transcriptomic analysis of human germinal center B cells\n    # Only 6 proteins are measured: ['CD184', 'CD44', 'CD69', 'CD83', 'CD9', 'CD196']\n    # 4513 cells GC3a\n    # 4777 cells:  GC3b \n\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc \nif dict_datasets2consider['GSE139891 HGC-Bcells'] == 'Yes':\n    for     key4dict in [ 'GSE139891 HGC-Bcells Part1', 'GSE139891 HGC-Bcells Part2']: \n    # GSE139891 Single-cell transcriptomic analysis of human germinal center B cells\n    # Only 6 proteins are measured: ['CD184', 'CD44', 'CD69', 'CD83', 'CD9', 'CD196']\n    # 4513 cells GC3a\n    # 4777 cells:  GC3b \n    \n        if work_with_protein_data == 'Yes':\n            if key4dict == 'GSE139891 HGC-Bcells Part1':\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560816_GC3a_Ab_counts.txt/GC3a_Ab_counts.txt' # Proteins (4513, 6)\n            else:\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560817_GC3b_Ab_counts.txt/GC3b_Ab_counts.txt' # proteins # 4777 cells:  GC3b\n\n            df = pd.read_csv(fn, index_col = 0 , sep = '\\t').T\n            print('Original shape:', df.shape)\n            df = df.iloc[:N_rows2take,:]\n            # print( df.info() )\n            l = [t.split('_')[1] for t in df.columns]\n            df.columns = l    \n            dict_df_prot[key4dict] = df\n            print('Protein data shape:', df.shape)\n            display(df.head(2)) \n\n        if work_with_rna_data == 'Yes':\n            if key4dict == 'GSE139891 HGC-Bcells Part1':\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560814_GC3a_umi.txt/counts_no_bad.txt' # RNA # (4513, 24019) \n            else:\n                fn = '/kaggle/input/citeseq-rnaseq/GSM4560815_GC3b_umi.txt/counts_no_bad.txt' # RNA 4777\n\n            df = pd.read_csv(fn, index_col = 0 , sep = '\\t').T\n            print('Original shape:', df.shape)\n            df = df.iloc[:N_rows2take,:]\n            l = [t.split(';')[1] for t in df.columns]\n            df.columns = l    \n            print('RNA data shape:', df.shape)\n            dict_df_rna[key4dict] = df\n            display(df.head(2)) \n\n\n        df = pd.DataFrame(index = df.index)\n        print('Meta data shape:', df.shape)\n        dict_df_meta[key4dict] = df\n        display(df.head(2)) \n\n        gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:45:07.644862Z","iopub.execute_input":"2023-03-14T21:45:07.645319Z","iopub.status.idle":"2023-03-14T21:46:06.582865Z","shell.execute_reply.started":"2023-03-14T21:45:07.645278Z","shell.execute_reply":"2023-03-14T21:46:06.581320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE160766 \n\n\tSingle-cell CITE-seq of two 9 month old BALBc mouse spleens\n    \nhttps://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE160766\n","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nif dict_datasets2consider['GSE160766 MouseSpleen'] == 'Yes':\n    # https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE160766\n    if work_with_protein_data == 'Yes':\n        key4dict = 'GSE160766 MouseSpleen'\n        fn = '/kaggle/input/cite-seqscrna-seq-proteins-gse160766/GSE160766_protein_expr_all.csv/GSE160766_protein_expr_all.csv'\n        df = pd.read_csv(fn).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        l = df.columns\n    #     print(l)\n        l2 = [t.split('_')[1] for t in l]\n        df.columns = l2\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2)) \n    \n    if work_with_rna_data == 'Yes':\n        fn = '/kaggle/input/cite-seqscrna-seq-proteins-gse160766/GSE160766_gene_matrix_all_new.csv/GSE160766_gene_matrix_all_new.csv'\n        df = pd.read_csv(fn).T\n        print('Original shape:', df.shape)\n        df = df.iloc[:N_rows2take,:]\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n\n    df = pd.DataFrame(index = df.index)\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:46:06.584295Z","iopub.execute_input":"2023-03-14T21:46:06.584644Z","iopub.status.idle":"2023-03-14T21:46:21.488415Z","shell.execute_reply.started":"2023-03-14T21:46:06.584609Z","shell.execute_reply":"2023-03-14T21:46:21.486914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GSE155673 PBMC Covid","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nimport scipy.io\n\ntmp = dict_datasets2consider['GSE155673 PBMC Covid'] \nif (tmp == 'Yes') or (isinstance(tmp,(int,float, np.number) ) and tmp > 0) : #   'Yes':\n    N_data2load = 1 # \n    if (isinstance(tmp,(int,float, np.number) ) and tmp > 0) : \n        N_data2load = dict_datasets2consider['GSE155673 PBMC Covid']\n    \n    fn = '/kaggle/input/d/cosjo11/citeseqscrnaseqproteins-challenge-neurips2021/GSE155673_features.tsv/features.tsv'\n    df_genes_info = pd.read_csv(fn, sep = '\\t', header = None)\n    print(df_genes_info.shape)\n    display(df_genes_info.head(2))\n\n    m = df_genes_info[2] == 'Antibody Capture'\n    rna_names = list( df_genes_info[1][~m] )\n    rna_names[:2]\n    prot_names = list( df_genes_info[1][m] )\n    print(len(prot_names), 'Original prot names:', prot_names )\n    #print(prot_names)\n    def chg_name(s):\n        r = s.split('--')[0].split('_')[0]\n        if s.startswith('Isotype-control'): r = s\n        if s ==  'CD197-CCR7--G043H7-TSA' : r = 'CD197'\n        return r\n    prot_names = [chg_name(t) for t in prot_names]\n    print(len(prot_names), 'Transformed prot names:', prot_names )\n\n\n    list_data_ids = ['cov01', 'cov02', 'cov03', 'cov04', 'cov07', 'cov08', 'cov09', 'cov10', 'cov11', 'cov12', 'cov17', 'cov18']\n    for ii, str_id in enumerate( list_data_ids[:N_data2load] ):\n        print(ii, str_id)\n        key4dict = 'GSE155673 PBMC Covid ' + str_id\n\n        fn = '/kaggle/input/d/cosjo11/citeseqscrnaseqproteins-challenge-neurips2021/GSE155673_'+str_id+ '_matrix.mtx/matrix.mtx'\n        mtx = scipy.io.mmread(fn).T\n        print('Original shape rna+prots:', mtx.shape, type(mtx) )\n\n        fn = '/kaggle/input/d/cosjo11/citeseqscrnaseqproteins-challenge-neurips2021/GSE155673_'+str_id+'_barcodes.tsv/barcodes.tsv'\n        d = pd.read_csv(fn,header = None)\n        d\n\n        if work_with_protein_data == 'Yes':\n            df = pd.DataFrame(mtx.todense()[:N_rows2take,-len(prot_names ):], index = d[0].iloc[:N_rows2take] , columns = prot_names )\n            print(df.shape)\n            display(df.head(2))\n            dict_df_prot[key4dict] = df\n\n        if work_with_rna_data == 'Yes':\n            df = pd.DataFrame(mtx.todense()[:N_rows2take,:len(rna_names )], index = d[0].iloc[:N_rows2take] , columns = rna_names )\n            print(df.shape)\n            display(df.head(2))\n            dict_df_rna[key4dict] = df\n\n\n        df = pd.DataFrame(index = df.index)\n        print('Meta data shape:', df.shape)\n        dict_df_meta[key4dict] = df\n        display(df.head(2)) \n\n        gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:46:21.491365Z","iopub.execute_input":"2023-03-14T21:46:21.492950Z","iopub.status.idle":"2023-03-14T21:47:15.031584Z","shell.execute_reply.started":"2023-03-14T21:46:21.492887Z","shell.execute_reply":"2023-03-14T21:47:15.030142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis","metadata":{}},{"cell_type":"code","source":"print(list(dict_df_prot) )","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:47:15.033337Z","iopub.execute_input":"2023-03-14T21:47:15.034750Z","iopub.status.idle":"2023-03-14T21:47:15.040669Z","shell.execute_reply.started":"2023-03-14T21:47:15.034707Z","shell.execute_reply":"2023-03-14T21:47:15.039309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ndf_stat = pd.DataFrame(); IX = - 1\n\nfor k in dict_df_meta:\n    IX += 1\n    df_stat.loc[IX,'Data'] = str(k)\n    if k in dict_df_prot.keys():\n        df = dict_df_prot[k]\n        df_stat.loc[IX,'Protein Data Shape'] = str(df.shape)\n        df_stat.loc[IX,'Protein Raw Counts'] = np.all( (df.sum()).astype(int) ==  df.sum() )\n        df_stat.loc[IX,'Protein %Zeros'] = np.round( (df == 0).sum().sum()/df.shape[0]/df.shape[1]*100, 1)\n    if k in dict_df_rna.keys():\n        df = dict_df_rna[k]\n        df_stat.loc[IX,'RNA Data Shape'] = str(df.shape)\n        df_stat.loc[IX,'RNA Raw Counts'] = np.all( (df.sum()).astype(int) ==  df.sum() )\n        df_stat.loc[IX,'RNA %Zeros'] = np.round( (df == 0).sum().sum()/df.shape[0]/df.shape[1]*100, 1)\n    df = dict_df_meta[k]\n    df_stat.loc[IX,'Meta Data Shape'] = str(df.shape)\ndf_stat    ","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:47:15.042321Z","iopub.execute_input":"2023-03-14T21:47:15.042871Z","iopub.status.idle":"2023-03-14T21:47:17.521405Z","shell.execute_reply.started":"2023-03-14T21:47:15.042835Z","shell.execute_reply":"2023-03-14T21:47:17.520194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"execution":{"iopub.status.busy":"2023-03-14T21:47:17.523454Z","iopub.execute_input":"2023-03-14T21:47:17.523934Z","iopub.status.idle":"2023-03-14T21:47:17.530317Z","shell.execute_reply.started":"2023-03-14T21:47:17.523881Z","shell.execute_reply":"2023-03-14T21:47:17.528890Z"},"trusted":true},"execution_count":null,"outputs":[]}]}