{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nLoad protein part for several CITE-seq data. \n\nData is stored in dict_df with a key - which contain info on a dataset.\n\nSome simple statistics is calculated. \n\n\n**Observations:**\n\n    CD45RA and CD45RO are typically negatively correlated and have different localizations (i.e. areas where they have high values). \n    For KaggleNIPS22 the correlation is rather low (-0.07) , but still the localization (on umap) is different !\n    \n    \n    Trimap in general is quite good, but if data contains outliers trimap will focus on them rather than dropping them neglecting them. As umap does. \n    Also Trimap crashes the RAM for some of our examples - that is strange. \n    So in later versions of the notebook ( > 4 ) we skip it. \n    \n    CD36 anticorrelate with CD45, CD45RA - for several datasets\n    CD44 correlate with CD45, CD45RA,CD45RO - for several datasets\n    СД71 в топах антикор с сд45\n    СД11 ведет себя по разному\n\nFor tests of different dim-red algorithms see e.g.: https://www.kaggle.com/code/alexandervc/cite-seq2023-dim-red-visualizations\n","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport time \nt0start = time.time()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:16.772695Z","iopub.execute_input":"2023-03-12T22:03:16.773475Z","iopub.status.idle":"2023-03-12T22:03:16.782201Z","shell.execute_reply.started":"2023-03-12T22:03:16.773433Z","shell.execute_reply":"2023-03-12T22:03:16.780415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-12T22:03:16.787086Z","iopub.execute_input":"2023-03-12T22:03:16.787937Z","iopub.status.idle":"2023-03-12T22:03:16.813158Z","shell.execute_reply.started":"2023-03-12T22:03:16.787900Z","shell.execute_reply":"2023-03-12T22:03:16.811350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scanpy\nimport scanpy as sc","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:16.815846Z","iopub.execute_input":"2023-03-12T22:03:16.816521Z","iopub.status.idle":"2023-03-12T22:03:28.533707Z","shell.execute_reply.started":"2023-03-12T22:03:16.816467Z","shell.execute_reply":"2023-03-12T22:03:28.532206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"dict_df = {}\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:28.536122Z","iopub.execute_input":"2023-03-12T22:03:28.536481Z","iopub.status.idle":"2023-03-12T22:03:28.542619Z","shell.execute_reply.started":"2023-03-12T22:03:28.536447Z","shell.execute_reply":"2023-03-12T22:03:28.541374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## NIPS22-Kaggle ","metadata":{}},{"cell_type":"code","source":"%%time\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\ndf = pd.read_hdf(FP_CITE_TRAIN_TARGETS)\nprint( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\ndict_df['NIPS2022'] = df\ndf","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:28.543867Z","iopub.execute_input":"2023-03-12T22:03:28.544164Z","iopub.status.idle":"2023-03-12T22:03:29.361627Z","shell.execute_reply.started":"2023-03-12T22:03:28.544136Z","shell.execute_reply":"2023-03-12T22:03:29.360096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## NIPS21","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/citeseqscrnaseqproteins-challenge-neurips2021/GSE194122_openproblems_neurips2021_cite_BMMC_processed.h5ad'\nadata = sc.read(fn)\nadata.var_names_make_unique()\nprint(adata)\nmask = adata.var['feature_types' ] == 'ADT'\nX = adata[:,mask].X.todense()\ndf = pd.DataFrame(X, columns = list(adata.var['feature_types' ][mask].index)  )\nl = [t.replace('-1','') for t in df.columns]\ndf.columns = l\nprint( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\nprint(df.shape)\ndict_df['NIPS2021'] = df\ndf","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:29.364411Z","iopub.execute_input":"2023-03-12T22:03:29.364768Z","iopub.status.idle":"2023-03-12T22:03:45.628551Z","shell.execute_reply.started":"2023-03-12T22:03:29.364734Z","shell.execute_reply":"2023-03-12T22:03:45.627287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kaggle 2302 community competition ","metadata":{}},{"cell_type":"code","source":"%%time\n#files_path = '/kaggle/input/machine-learning-challenge-2-prediction/'\ndf_X = pd.read_csv('/kaggle/input/machine-learning-challenge-2-prediction/training_set_rna.csv', index_col = 0).T    # Train features \n#df = df_X # Will simple denote it df\ndf_Y = pd.read_csv('/kaggle/input/machine-learning-challenge-2-prediction/training_set_adt.csv', index_col = 0).T # Train targets     \ndf_X_submission = pd.read_csv('/kaggle/input/machine-learning-challenge-2-prediction/test_set_rna.csv', index_col = 0).T # Features to prepare submission \n\nprint(df_X.shape, df_Y.shape, df_X_submission.shape)\ndisplay(df_X.head(2))\ndisplay(df_Y.head(2))\ndf = df_Y\nprint( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n\ndict_df['Kaggle2302'] = df\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:45.630088Z","iopub.execute_input":"2023-03-12T22:03:45.630574Z","iopub.status.idle":"2023-03-12T22:03:47.076024Z","shell.execute_reply.started":"2023-03-12T22:03:45.630517Z","shell.execute_reply":"2023-03-12T22:03:47.074685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## GSE148127","metadata":{}},{"cell_type":"code","source":"\nfn = '/kaggle/input/citeseqgse148127/GSE148127_ADT.counts.csv' # No CD53\ndf = pd.read_csv(fn,index_col = 0).T\nprint('Original column names', df.columns)\ndef CDname_change(s):\n    r = s.split('-')[1]\n    if s == 'CITE-CCR2-CD192': r = 'CD192'\n    if s == 'CITE-PD-1': r = 'PD-1'\n    if s == 'CITE-PD-L1': r = 'PD-L1'\n    if s == 'CITE-I-A-I-E': r = 'I-A-I-E'\n    return r\nl = [CDname_change(t) for t in df.columns]\ndf.columns = l\nprint('Processed column names', df.columns)\n\ndict_df['GSE148127 Mouse'] = df\nprint( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\ndf","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:47.077504Z","iopub.execute_input":"2023-03-12T22:03:47.077911Z","iopub.status.idle":"2023-03-12T22:03:47.560184Z","shell.execute_reply.started":"2023-03-12T22:03:47.077879Z","shell.execute_reply":"2023-03-12T22:03:47.558988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analysis","metadata":{}},{"cell_type":"code","source":"print(list(dict_df) )","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:47.561667Z","iopub.execute_input":"2023-03-12T22:03:47.562096Z","iopub.status.idle":"2023-03-12T22:03:47.567549Z","shell.execute_reply.started":"2023-03-12T22:03:47.562062Z","shell.execute_reply":"2023-03-12T22:03:47.566678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlations","metadata":{}},{"cell_type":"code","source":"df_stat = pd.DataFrame(); IX = -1\ncol1 = 'CD45RA'; col2 = 'CD45RO'\nfor k in dict_df:\n    df = dict_df[k]\n    if ( col1 in df.columns) and ( col2 in df.columns):\n        c = np.corrcoef(df[col1], df[col2])[0,1]\n        IX+=1\n        df_stat.loc[IX,'Data'] = k \n        df_stat.loc[IX,'Corr '+col1+' '+col2] = c \n        df_stat.loc[IX,'Std '+col1] = df[col1].std() \n        df_stat.loc[IX,'Mean '+col1] = df[col1].mean() \n        df_stat.loc[IX,'Std '+col2] = df[col2].std() \n        df_stat.loc[IX,'Mean '+col2] = df[col2].mean() \n        \n\ndf_stat.to_csv('corr_stat_'+col1+'_'+col2+'.csv')        \ndf_stat        ","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:47.569222Z","iopub.execute_input":"2023-03-12T22:03:47.570169Z","iopub.status.idle":"2023-03-12T22:03:47.631701Z","shell.execute_reply.started":"2023-03-12T22:03:47.570125Z","shell.execute_reply":"2023-03-12T22:03:47.630021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pair_cors(col1,col2):\n    df_stat = pd.DataFrame(); IX = -1\n#     col1 = 'CD45RA'; col2 = 'CD45RO'\n    for k in dict_df:\n        df = dict_df[k]\n        if ( col1 in df.columns) and ( col2 in df.columns):\n            c = np.corrcoef(df[col1], df[col2])[0,1]\n            IX+=1\n            df_stat.loc[IX,'Data'] = k \n            df_stat.loc[IX,'Corr '+col1+' '+col2] = c \n            df_stat.loc[IX,'Std '+col1] = df[col1].std() \n            df_stat.loc[IX,'Mean '+col1] = df[col1].mean() \n            df_stat.loc[IX,'Std '+col2] = df[col2].std() \n            df_stat.loc[IX,'Mean '+col2] = df[col2].mean() \n\n\n    df_stat.to_csv('corr_stat_'+col1+'_'+col2+'.csv')        \n    display(df_stat )\n    return df_stat\npair_cors(col1 = 'CD45',col2 = 'CD44')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:47.633508Z","iopub.execute_input":"2023-03-12T22:03:47.634061Z","iopub.status.idle":"2023-03-12T22:03:47.705954Z","shell.execute_reply.started":"2023-03-12T22:03:47.634003Z","shell.execute_reply":"2023-03-12T22:03:47.704604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from itertools import combinations\ndf_stat = pd.DataFrame(); IX = -1\nlist_names = ['CD45RA' ,  'CD45RO', 'CD45' ] \ncombos = list(combinations(list_names, 2))\ncombos\nfor k in dict_df:\n    df = dict_df[k]\n    IX+=1\n    df_stat.loc[IX,'Data'] = k \n    for pairs in combos:\n        col1 = pairs[0]; col2 = pairs[1]\n        if ( col1 in df.columns) and ( col2 in df.columns):\n            c = np.corrcoef(df[col1], df[col2])[0,1]\n            df_stat.loc[IX,'Corr '+col1+' '+col2] = c\n    for col in list_names:\n        if col not in df.columns: continue\n        df_stat.loc[IX,'Std '+col] = df[col].std() \n        df_stat.loc[IX,'Mean '+col] = df[col].mean() \n        \ndf_stat.to_csv('corr_stat_several_protein_pairs.csv')        \n\ndf_stat ","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:47.709799Z","iopub.execute_input":"2023-03-12T22:03:47.710275Z","iopub.status.idle":"2023-03-12T22:03:47.785932Z","shell.execute_reply.started":"2023-03-12T22:03:47.710198Z","shell.execute_reply":"2023-03-12T22:03:47.784589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scatterplots","metadata":{}},{"cell_type":"code","source":"col1 = 'CD45RA'; col2 = 'CD45RO'\nfor k in dict_df:\n    df = dict_df[k]\n    if ( col1 in df.columns) and ( col2 in df.columns):\n        c = np.corrcoef(df[col1], df[col2])[0,1]\n\n        plt.figure(figsize= (20,10))\n        sns.scatterplot(x=df[col1], y = df[col2] )\n        plt.title(k + ' Corr = '+str(np.round(c,2)), fontsize = 20)\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:47.787843Z","iopub.execute_input":"2023-03-12T22:03:47.788570Z","iopub.status.idle":"2023-03-12T22:03:49.299091Z","shell.execute_reply.started":"2023-03-12T22:03:47.788523Z","shell.execute_reply":"2023-03-12T22:03:49.297677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 1000)\npd.set_option('max_colwidth', -1)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:49.301031Z","iopub.execute_input":"2023-03-12T22:03:49.301516Z","iopub.status.idle":"2023-03-12T22:03:49.307655Z","shell.execute_reply.started":"2023-03-12T22:03:49.301477Z","shell.execute_reply":"2023-03-12T22:03:49.306655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nlist_names = ['CD45RA' ,  'CD45RO', 'CD45' ] \ndf_stat = pd.DataFrame(); IX = -1\nfor k in dict_df:\n    df = dict_df[k]\n    IX+=1\n    df_stat.loc[IX,'Data'] = k \n    for col1 in list_names:\n        if col1 not in df.columns: continue \n        d = {}\n        for col2 in df.columns:\n            c = np.corrcoef(df[col1], df[col2])[0,1]\n            d[col2] = c\n        s = pd.Series(d) \n        s = s.sort_values(ascending = False).round(2)\n        ss = str(dict(s.iloc[1:15]))\n        df_stat.loc[IX,col1 +' TopCorr'] = ss\n        ss = str(dict(s.tail(5)))\n        df_stat.loc[IX,col1 +' TailCorr'] = ss\n        \n\ndf_stat.to_csv('top_correlations_to_selected_proteins.csv') \n        \ndf_stat \n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:49.308877Z","iopub.execute_input":"2023-03-12T22:03:49.309866Z","iopub.status.idle":"2023-03-12T22:03:50.930289Z","shell.execute_reply.started":"2023-03-12T22:03:49.309829Z","shell.execute_reply":"2023-03-12T22:03:50.929013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = pair_cors(col1 = 'CD45',col2 = 'CD44')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:50.931803Z","iopub.execute_input":"2023-03-12T22:03:50.932155Z","iopub.status.idle":"2023-03-12T22:03:50.976591Z","shell.execute_reply.started":"2023-03-12T22:03:50.932121Z","shell.execute_reply":"2023-03-12T22:03:50.975370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# UMAP","metadata":{}},{"cell_type":"code","source":"import umap\n\nreducer = umap.UMAP(random_state = 42)\n\nimport gc \ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:50.980073Z","iopub.execute_input":"2023-03-12T22:03:50.980419Z","iopub.status.idle":"2023-03-12T22:03:52.146033Z","shell.execute_reply.started":"2023-03-12T22:03:50.980387Z","shell.execute_reply":"2023-03-12T22:03:52.144523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport gc\n# col1 = 'CD45RA'; col2 = 'CD45RO'\nlist4color = ['CD45RA','CD45RO']\nfor k in dict_df:\n    df = dict_df[k]\n    r = reducer.fit_transform(df)\n    fig = plt.figure(figsize= (20,10))\n    plt.suptitle(k, fontsize = 20 )\n    \n    c = 0\n    for col in list4color: \n        col_loc = col\n        if col not in df.columns:\n            col_loc = df.columns[c]\n            if k ==  'GSE148127 Mouse':\n                if (c == 0): col_loc = 'CD45'# 'CITE-CD45'\n                if (c == 1): col_loc =  'CD45R'# 'CITE-CD45R-B220'\n                \n        v = df[col_loc]# df.columns[c]]\n        v = np.clip(v, np.percentile(v,5), np.percentile(v,95) )\n        \n        c += 1\n        fig.add_subplot(1,len(list4color),c)\n        sns.scatterplot(x=r[:,0], y = r[:,1], hue = v , palette = 'rainbow')\n        plt.title( col_loc , fontsize = 20)\n    \n    gc.collect()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:03:52.148360Z","iopub.execute_input":"2023-03-12T22:03:52.148704Z","iopub.status.idle":"2023-03-12T22:07:32.490171Z","shell.execute_reply.started":"2023-03-12T22:03:52.148671Z","shell.execute_reply":"2023-03-12T22:07:32.488970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function to make plots","metadata":{}},{"cell_type":"code","source":"%%time\nimport gc\nimport umap \n\ndef plots01(list4color ):\n    # col1 = 'CD45RA'; col2 = 'CD45RO'\n    #list4color = ['CD45RA','CD45RO','CD45']\n    for k in dict_df:\n        df = dict_df[k]\n        reducer = umap.UMAP(random_state = 42 ) \n        \n        r = reducer.fit_transform(df)\n        fig = plt.figure(figsize= (20,10))\n        plt.suptitle(k, fontsize = 20 )\n\n        c = 0\n        for col in list4color: \n            col_loc = col\n            if col not in df.columns:\n                col_loc = df.columns[c]\n                if k ==  'GSE148127 Mouse':\n                    if (c == 0): col_loc = 'CD45'# 'CITE-CD45'\n                    if (c == 1): col_loc =  'CD45R'# 'CITE-CD45R-B220'\n\n            v = df[col_loc]# df.columns[c]]\n            v = np.clip(v, np.percentile(v,5), np.percentile(v,95) )\n\n            c += 1\n            fig.add_subplot(1,len(list4color),c)\n            sns.scatterplot(x=r[:,0], y = r[:,1], hue = v , palette = 'rainbow')\n            plt.title( col_loc , fontsize = 20)\n\n        plt.show()\n        gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:07:32.491543Z","iopub.execute_input":"2023-03-12T22:07:32.491860Z","iopub.status.idle":"2023-03-12T22:07:32.502196Z","shell.execute_reply.started":"2023-03-12T22:07:32.491829Z","shell.execute_reply":"2023-03-12T22:07:32.501010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Color by several proteins ","metadata":{}},{"cell_type":"code","source":"%%time\nlist4color = ['CD45RA','CD45RO','CD45','CD3','CD4']# , 'MALAT1', 'NEAT1']\nplots01(list4color )","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:07:32.503730Z","iopub.execute_input":"2023-03-12T22:07:32.504050Z","iopub.status.idle":"2023-03-12T22:11:43.655812Z","shell.execute_reply.started":"2023-03-12T22:07:32.504019Z","shell.execute_reply":"2023-03-12T22:11:43.654532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nlist4color = ['CD45RA','CD45RO','CD45','CD36','CD71']# , 'MALAT1', 'NEAT1']\nplots01(list4color )","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:11:43.658355Z","iopub.execute_input":"2023-03-12T22:11:43.658784Z","iopub.status.idle":"2023-03-12T22:15:56.568920Z","shell.execute_reply.started":"2023-03-12T22:11:43.658744Z","shell.execute_reply":"2023-03-12T22:15:56.567563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nlist4color = ['CD45RA','CD45RO','CD45','CD44']# , 'MALAT1', 'NEAT1']\nplots01(list4color )","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:15:56.572042Z","iopub.execute_input":"2023-03-12T22:15:56.572554Z","iopub.status.idle":"2023-03-12T22:20:00.969529Z","shell.execute_reply.started":"2023-03-12T22:15:56.572506Z","shell.execute_reply":"2023-03-12T22:20:00.968192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nlist4color = ['CD45RA','CD45RO','CD45','CD34']# , 'MALAT1', 'NEAT1']\nplots01(list4color )","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:20:12.681127Z","iopub.execute_input":"2023-03-12T22:20:12.682515Z","iopub.status.idle":"2023-03-12T22:24:18.953901Z","shell.execute_reply.started":"2023-03-12T22:20:12.682476Z","shell.execute_reply":"2023-03-12T22:24:18.952574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Trimap","metadata":{}},{"cell_type":"code","source":"!pip install trimap\nimport trimap\n\nreducer = trimap.TRIMAP()# UMAP(random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:20:00.973020Z","iopub.execute_input":"2023-03-12T22:20:00.974255Z","iopub.status.idle":"2023-03-12T22:20:12.630169Z","shell.execute_reply.started":"2023-03-12T22:20:00.974179Z","shell.execute_reply":"2023-03-12T22:20:12.627937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Trimap causes crash by RAM ')","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:20:12.633123Z","iopub.execute_input":"2023-03-12T22:20:12.633640Z","iopub.status.idle":"2023-03-12T22:20:12.642206Z","shell.execute_reply.started":"2023-03-12T22:20:12.633589Z","shell.execute_reply":"2023-03-12T22:20:12.640401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport gc\nif 1:\n    print('Trimap causes crash by RAM so we skip it ')\nelse:\n    # col1 = 'CD45RA'; col2 = 'CD45RO'\n    list4color = ['CD45RA','CD45RO']\n    for k in dict_df:\n        df = dict_df[k]\n        \n        reducer = trimap.TRIMAP()# UMAP(random_state = 42)\n        r = reducer.fit_transform(df)\n        fig = plt.figure(figsize= (20,10))\n        plt.suptitle(k, fontsize = 20 )\n\n        c = 0\n        for col in list4color: \n            col_loc = col\n            if col not in df.columns:\n                col_loc = df.columns[c]\n                if k ==  'GSE148127 Mouse':\n                    if (c == 0): col_loc =  'CITE-CD45'\n                    if (c == 1): col_loc =  'CITE-CD45R-B220'\n\n            v = df[col_loc]# df.columns[c]]\n            v = np.clip(v, np.percentile(v,5), np.percentile(v,95) )\n\n            c += 1\n            fig.add_subplot(1,len(list4color),c)\n            sns.scatterplot(x=r[:,0], y = r[:,1], hue = v , palette = 'rainbow')\n            plt.title(k + '  ' + col_loc , fontsize = 20)\n\n        gc.collect()\n        plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:20:12.644762Z","iopub.execute_input":"2023-03-12T22:20:12.645985Z","iopub.status.idle":"2023-03-12T22:20:12.659820Z","shell.execute_reply.started":"2023-03-12T22:20:12.645932Z","shell.execute_reply":"2023-03-12T22:20:12.657639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"execution":{"iopub.status.busy":"2023-03-12T22:20:12.662137Z","iopub.execute_input":"2023-03-12T22:20:12.663075Z","iopub.status.idle":"2023-03-12T22:20:12.676439Z","shell.execute_reply.started":"2023-03-12T22:20:12.663013Z","shell.execute_reply":"2023-03-12T22:20:12.674717Z"},"trusted":true},"execution_count":null,"outputs":[]}]}