{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-02T21:41:02.007106Z","iopub.execute_input":"2023-01-02T21:41:02.007682Z","iopub.status.idle":"2023-01-02T21:41:02.081697Z","shell.execute_reply.started":"2023-01-02T21:41:02.007543Z","shell.execute_reply":"2023-01-02T21:41:02.080346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fn = '/kaggle/input/research-project-01-around-multimodal-singlecell/CD_info.csv'\nfn = '/kaggle/input/research-project-01-around-multimodal-singlecell/CD_to_EnsemblSymbol_correspondence_NIPS2022.csv'\ndf = pd.read_csv(fn,index_col = 0)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.083846Z","iopub.execute_input":"2023-01-02T21:41:02.085042Z","iopub.status.idle":"2023-01-02T21:41:02.147738Z","shell.execute_reply.started":"2023-01-02T21:41:02.084991Z","shell.execute_reply":"2023-01-02T21:41:02.146348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(50)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.150222Z","iopub.execute_input":"2023-01-02T21:41:02.150759Z","iopub.status.idle":"2023-01-02T21:41:02.185087Z","shell.execute_reply.started":"2023-01-02T21:41:02.150697Z","shell.execute_reply":"2023-01-02T21:41:02.183722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.tail(50)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.206030Z","iopub.execute_input":"2023-01-02T21:41:02.206470Z","iopub.status.idle":"2023-01-02T21:41:02.239031Z","shell.execute_reply.started":"2023-01-02T21:41:02.206431Z","shell.execute_reply":"2023-01-02T21:41:02.237690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for kw in ['ligand']:# 'adhesion', 'phosphatase ', 'integrin', 'immunoglobulin',  'interleukin', ' lectin', 'chemokine', 'selectin' ]:\n    print(); print(kw)\n    for I in df.index:\n        if kw  in str(df.loc[I,'name']):\n            print(df.loc[I,'CD_name'], df.loc[I,'name'])\n","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.241420Z","iopub.execute_input":"2023-01-02T21:41:02.242192Z","iopub.status.idle":"2023-01-02T21:41:02.254721Z","shell.execute_reply.started":"2023-01-02T21:41:02.242145Z","shell.execute_reply":"2023-01-02T21:41:02.253464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 1000)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.256383Z","iopub.execute_input":"2023-01-02T21:41:02.257076Z","iopub.status.idle":"2023-01-02T21:41:02.265390Z","shell.execute_reply.started":"2023-01-02T21:41:02.256839Z","shell.execute_reply":"2023-01-02T21:41:02.263891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.294184Z","iopub.execute_input":"2023-01-02T21:41:02.294608Z","iopub.status.idle":"2023-01-02T21:41:02.366995Z","shell.execute_reply.started":"2023-01-02T21:41:02.294574Z","shell.execute_reply":"2023-01-02T21:41:02.365571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Check that symbol used in NIPS22 data is the same as in the symbol obtained by \"mygene\" (column \"symbol\") ')\nfor i in range(140):\n    t1 = df['ID in RNA dataset'].iat[i]\n    t2 = df['symbol'].iat[i]\n    if pd.notnull(t1):\n        t1 = t1.split('_')[1]\n        if t1 != t2:\n            print(i,t1,t2)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.368404Z","iopub.execute_input":"2023-01-02T21:41:02.368944Z","iopub.status.idle":"2023-01-02T21:41:02.379593Z","shell.execute_reply.started":"2023-01-02T21:41:02.368906Z","shell.execute_reply":"2023-01-02T21:41:02.378179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.reset_index()\ndf.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.413955Z","iopub.execute_input":"2023-01-02T21:41:02.414510Z","iopub.status.idle":"2023-01-02T21:41:02.437284Z","shell.execute_reply.started":"2023-01-02T21:41:02.414451Z","shell.execute_reply":"2023-01-02T21:41:02.435980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(['Description'], axis = 1) # , 'ID in RNA dataset'\ndf.head(1)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.497675Z","iopub.execute_input":"2023-01-02T21:41:02.498114Z","iopub.status.idle":"2023-01-02T21:41:02.516338Z","shell.execute_reply.started":"2023-01-02T21:41:02.498080Z","shell.execute_reply":"2023-01-02T21:41:02.514965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.540947Z","iopub.execute_input":"2023-01-02T21:41:02.541442Z","iopub.status.idle":"2023-01-02T21:41:02.549338Z","shell.execute_reply.started":"2023-01-02T21:41:02.541324Z","shell.execute_reply":"2023-01-02T21:41:02.547859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns = ['index in NIPS22 targets', 'Ensembl ID', 'CD_name', 'In NIPS22 RNA dataset', 'ID in NIPS22 RNA dataset', 'symbol', 'entrezgene', 'brief info', 'alias']\ndf.head(1)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.599028Z","iopub.execute_input":"2023-01-02T21:41:02.600772Z","iopub.status.idle":"2023-01-02T21:41:02.618544Z","shell.execute_reply.started":"2023-01-02T21:41:02.600712Z","shell.execute_reply":"2023-01-02T21:41:02.617347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Search for keywords - 'adhesion', 'phosphatase ', 'integrin', 'immunoglobulin' ... and add such information ","metadata":{}},{"cell_type":"code","source":"col_name4inf = 'brief info'\ndf['keyword'] = np.nan\nfor kw in ['adhesion', 'phosphatase ', 'integrin', 'immunoglobulin',  'interleukin', ' lectin', 'chemokine', 'selectin', 'ligand' ]:\n    print(); print(kw)\n    for I in df.index:\n        if kw  in str(df.loc[I,col_name4inf]):\n            print(df.loc[I,'CD_name'], df.loc[I,col_name4inf])\n            if kw != 'ligand':\n                df.loc[I,'keyword'] = kw\n            else:\n                df.loc[I,'keyword'] = df.loc[I,col_name4inf]\ndf.sort_values('keyword').head(50)            ","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.658911Z","iopub.execute_input":"2023-01-02T21:41:02.659350Z","iopub.status.idle":"2023-01-02T21:41:02.749823Z","shell.execute_reply.started":"2023-01-02T21:41:02.659310Z","shell.execute_reply":"2023-01-02T21:41:02.748571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.751966Z","iopub.execute_input":"2023-01-02T21:41:02.752344Z","iopub.status.idle":"2023-01-02T21:41:02.760998Z","shell.execute_reply.started":"2023-01-02T21:41:02.752302Z","shell.execute_reply":"2023-01-02T21:41:02.759242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =df[ ['CD_name','symbol', 'Ensembl ID','entrezgene', 'alias', 'index in NIPS22 targets',   \n         'In NIPS22 RNA dataset', 'ID in NIPS22 RNA dataset', 'keyword',  'brief info' ] ]\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.762652Z","iopub.execute_input":"2023-01-02T21:41:02.763824Z","iopub.status.idle":"2023-01-02T21:41:02.787263Z","shell.execute_reply.started":"2023-01-02T21:41:02.763780Z","shell.execute_reply":"2023-01-02T21:41:02.785725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in [83,122,123 ]:\n    print( df.loc[i,'CD_name'])\n    print( df.loc[i,'keyword'])\n    ","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.790128Z","iopub.execute_input":"2023-01-02T21:41:02.790527Z","iopub.status.idle":"2023-01-02T21:41:02.802048Z","shell.execute_reply.started":"2023-01-02T21:41:02.790489Z","shell.execute_reply":"2023-01-02T21:41:02.800463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.loc[83,'brief info'] = 'T cell receptor'\ndf.loc[122,'brief info'] = 'T cell receptor'\ndf.loc[123,'brief info'] = 'T cell receptor'\n\ndf.loc[83,'keyword'] = 'T cell receptor'\ndf.loc[122,'keyword'] = 'T cell receptor'\ndf.loc[123,'keyword'] = 'T cell receptor'","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.803482Z","iopub.execute_input":"2023-01-02T21:41:02.803992Z","iopub.status.idle":"2023-01-02T21:41:02.814017Z","shell.execute_reply.started":"2023-01-02T21:41:02.803935Z","shell.execute_reply":"2023-01-02T21:41:02.812992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.816682Z","iopub.execute_input":"2023-01-02T21:41:02.817053Z","iopub.status.idle":"2023-01-02T21:41:02.878987Z","shell.execute_reply.started":"2023-01-02T21:41:02.817016Z","shell.execute_reply":"2023-01-02T21:41:02.877472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add information on protein levels from Kaggle-NIPS22 dataset","metadata":{}},{"cell_type":"code","source":"%%time\ndf_y = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ndisplay(df_y)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:02.881193Z","iopub.execute_input":"2023-01-02T21:41:02.881559Z","iopub.status.idle":"2023-01-02T21:41:03.985664Z","shell.execute_reply.started":"2023-01-02T21:41:02.881524Z","shell.execute_reply":"2023-01-02T21:41:03.984369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(df_y.columns) == set(df['CD_name'])","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:03.987396Z","iopub.execute_input":"2023-01-02T21:41:03.988212Z","iopub.status.idle":"2023-01-02T21:41:03.996146Z","shell.execute_reply.started":"2023-01-02T21:41:03.988168Z","shell.execute_reply":"2023-01-02T21:41:03.994787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = df_y.mean(axis = 0)\nt.name = 'Protein mean NIPS22'\ndf = pd.merge(df, t, left_on = 'CD_name', right_index= True ) # .join(t)\n\nt = df_y.std(axis = 0)\nt.name = 'Protein std NIPS22'\n# df_stat2 = df_stat2.join(t)\ndf = pd.merge(df, t, left_on = 'CD_name', right_index= True ) # .join(t)\n\n# # t = df_y.var(axis = 0)\n# # t.name = 'Protein var'\n# # df_stat2 = df_stat2.join(t)\n\n# df_stat2\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:03.999696Z","iopub.execute_input":"2023-01-02T21:41:04.000289Z","iopub.status.idle":"2023-01-02T21:41:04.167954Z","shell.execute_reply.started":"2023-01-02T21:41:04.000240Z","shell.execute_reply":"2023-01-02T21:41:04.166872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = '/kaggle/input/open-problems-multimodal/metadata.csv'\ndf_meta = pd.read_csv(fn, index_col = 0 )\ndf_meta\n# Cut only train cite-seq part: \nd = pd.DataFrame(index = df_y.index)\nprint(d.shape)\ndf_meta = d.join(df_meta, how = 'left')\ndf_meta","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:04.169444Z","iopub.execute_input":"2023-01-02T21:41:04.170112Z","iopub.status.idle":"2023-01-02T21:41:04.758405Z","shell.execute_reply.started":"2023-01-02T21:41:04.170071Z","shell.execute_reply":"2023-01-02T21:41:04.757276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d2 = df_meta.join(df_y)\nd2.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:04.759922Z","iopub.execute_input":"2023-01-02T21:41:04.760547Z","iopub.status.idle":"2023-01-02T21:41:04.876230Z","shell.execute_reply.started":"2023-01-02T21:41:04.760504Z","shell.execute_reply":"2023-01-02T21:41:04.874910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# d2 = df_meta.join(df_y)\n# d2.head(2)\nd3 = d2.groupby('cell_type').mean()\nd3 = d3.drop('day', axis = 1)\nd3 = d3.drop('donor', axis = 1)\nd3","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:04.877612Z","iopub.execute_input":"2023-01-02T21:41:04.877988Z","iopub.status.idle":"2023-01-02T21:41:05.101050Z","shell.execute_reply.started":"2023-01-02T21:41:04.877954Z","shell.execute_reply":"2023-01-02T21:41:05.099934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Protein SubTop Cell Type NIPS22'] = ''\nfor i,col in enumerate( df['CD_name']):# .columns:\n    ct = d3[col].sort_values(ascending = False).index[1]\n    IX = df.index[i]\n    df.loc[IX,'Protein SubTop Cell Type NIPS22'] = ct# d3.index[I]\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:05.102403Z","iopub.execute_input":"2023-01-02T21:41:05.102786Z","iopub.status.idle":"2023-01-02T21:41:05.197298Z","shell.execute_reply.started":"2023-01-02T21:41:05.102751Z","shell.execute_reply":"2023-01-02T21:41:05.195724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Protein Top Cell Type NIPS22'] = ''\nfor i,col in enumerate( df['CD_name']):# .columns:\n    I = d3[col].argmax()\n    IX = df.index[i]\n    df.loc[IX,'Protein Top Cell Type NIPS22'] = d3.index[I]\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:05.200497Z","iopub.execute_input":"2023-01-02T21:41:05.200978Z","iopub.status.idle":"2023-01-02T21:41:05.281045Z","shell.execute_reply.started":"2023-01-02T21:41:05.200941Z","shell.execute_reply":"2023-01-02T21:41:05.279711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Protein Top Cell Type NIPS22'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:05.282836Z","iopub.execute_input":"2023-01-02T21:41:05.283707Z","iopub.status.idle":"2023-01-02T21:41:05.292762Z","shell.execute_reply.started":"2023-01-02T21:41:05.283664Z","shell.execute_reply":"2023-01-02T21:41:05.291360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Protein SubTop Cell Type NIPS22'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:05.294594Z","iopub.execute_input":"2023-01-02T21:41:05.295050Z","iopub.status.idle":"2023-01-02T21:41:05.308481Z","shell.execute_reply.started":"2023-01-02T21:41:05.295003Z","shell.execute_reply":"2023-01-02T21:41:05.307170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ncm = df_y.corr()\ncm.head(5)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:05.310251Z","iopub.execute_input":"2023-01-02T21:41:05.312503Z","iopub.status.idle":"2023-01-02T21:41:09.292038Z","shell.execute_reply.started":"2023-01-02T21:41:05.312352Z","shell.execute_reply":"2023-01-02T21:41:09.290700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(df)): # , col in enumerate( df['CD_name']):# .columns:\n    IX = df.index[i]\n    cd = df['CD_name'].iat[i]\n    \n    v = cm[cd].sort_values(ascending = False)\n    \n    s = [v.index[j]+' (' + str(np.round(v.iat[j],2) ) + ')' for j in range(1,6)]\n    df.loc[IX,'Top5 Corr Proteins NIPS22'] = str(s)\n    \n    s = [v.index[j]+' (' + str(np.round(v.iat[j],2) ) + ')' for j in [-1,-2,-3,-4,-5] ]\n    df.loc[IX,'Top5 AntiCorr Proteins NIPS22'] = str(s)\n    \n    # print(cd, str(s))\ndf.head(20)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:09.294289Z","iopub.execute_input":"2023-01-02T21:41:09.294838Z","iopub.status.idle":"2023-01-02T21:41:09.484015Z","shell.execute_reply.started":"2023-01-02T21:41:09.294782Z","shell.execute_reply":"2023-01-02T21:41:09.482825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add information on RNA levels from Kaggle-NIPS22 dataset","metadata":{}},{"cell_type":"code","source":"%%time\ndf_rna = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_inputs.h5')\ndisplay(df_rna) ","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:41:09.485775Z","iopub.execute_input":"2023-01-02T21:41:09.486214Z","iopub.status.idle":"2023-01-02T21:42:09.875401Z","shell.execute_reply.started":"2023-01-02T21:41:09.486174Z","shell.execute_reply":"2023-01-02T21:42:09.874009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:42:09.876943Z","iopub.execute_input":"2023-01-02T21:42:09.877333Z","iopub.status.idle":"2023-01-02T21:42:09.884736Z","shell.execute_reply.started":"2023-01-02T21:42:09.877291Z","shell.execute_reply":"2023-01-02T21:42:09.883560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(df)): # , col in enumerate( df['CD_name']):# .columns:\n    IX = df.index[i]\n    id1 = df['ID in NIPS22 RNA dataset'].iat[i]\n    if pd.isnull(id1): continue\n    df.loc[IX,'RNA mean NIPS22'] = df_rna[id1].mean()\n    df.loc[IX,'RNA std NIPS22'] = df_rna[id1].std()\n    \ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:42:09.886013Z","iopub.execute_input":"2023-01-02T21:42:09.886369Z","iopub.status.idle":"2023-01-02T21:42:10.113889Z","shell.execute_reply.started":"2023-01-02T21:42:09.886338Z","shell.execute_reply":"2023-01-02T21:42:10.112581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = df['ID in NIPS22 RNA dataset'].notnull()\nl = df[m]['ID in NIPS22 RNA dataset'].tolist()\nl = list(set(l))\nprint(l)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:48:22.378726Z","iopub.execute_input":"2023-01-02T21:48:22.379138Z","iopub.status.idle":"2023-01-02T21:48:22.389972Z","shell.execute_reply.started":"2023-01-02T21:48:22.379105Z","shell.execute_reply":"2023-01-02T21:48:22.388393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nd2 = df_meta.join(df_rna[l])\n#d2.head(2)\nd3 = d2.groupby('cell_type').mean()\nd3 = d3.drop('day', axis = 1)\nd3 = d3.drop('donor', axis = 1)\nd3","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:48:25.655284Z","iopub.execute_input":"2023-01-02T21:48:25.655698Z","iopub.status.idle":"2023-01-02T21:48:25.901111Z","shell.execute_reply.started":"2023-01-02T21:48:25.655664Z","shell.execute_reply":"2023-01-02T21:48:25.899508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(df.shape[0]): #  ,col in enumerate( df['CD_name']):# .columns:\n    IX = df.index[i]\n    col = df[ 'ID in NIPS22 RNA dataset' ].iat[i]\n    if pd.isnull(col): continue \n    \n    ct = d3[col].sort_values(ascending = False).index[0]    \n    df.loc[IX,'RNA Top Cell Type NIPS22'] = ct# d3.index[I]\n    \n    ct = d3[col].sort_values(ascending = False).index[1]    \n    df.loc[IX,'RNA SubTop Cell Type NIPS22'] = ct# d3.index[I]\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:48:28.251180Z","iopub.execute_input":"2023-01-02T21:48:28.251585Z","iopub.status.idle":"2023-01-02T21:48:28.397270Z","shell.execute_reply.started":"2023-01-02T21:48:28.251550Z","shell.execute_reply":"2023-01-02T21:48:28.396072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m =df['RNA Top Cell Type NIPS22'] == df['Protein Top Cell Type NIPS22']\nm.sum()","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:50:17.965686Z","iopub.execute_input":"2023-01-02T21:50:17.966403Z","iopub.status.idle":"2023-01-02T21:50:17.974945Z","shell.execute_reply.started":"2023-01-02T21:50:17.966350Z","shell.execute_reply":"2023-01-02T21:50:17.973688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['RNA Top Cell Type NIPS22'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:50:37.710723Z","iopub.execute_input":"2023-01-02T21:50:37.712271Z","iopub.status.idle":"2023-01-02T21:50:37.727241Z","shell.execute_reply.started":"2023-01-02T21:50:37.712202Z","shell.execute_reply":"2023-01-02T21:50:37.725486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['RNA SubTop Cell Type NIPS22'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:51:27.328879Z","iopub.execute_input":"2023-01-02T21:51:27.329351Z","iopub.status.idle":"2023-01-02T21:51:27.339773Z","shell.execute_reply.started":"2023-01-02T21:51:27.329314Z","shell.execute_reply":"2023-01-02T21:51:27.338701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt \nsns.scatterplot(x = df['RNA mean NIPS22'], y=df['RNA std NIPS22'])# *df['std NIPS22'] )\nplt.show()\nsns.scatterplot(x = df['RNA mean NIPS22'], y=df['RNA std NIPS22']**2)# *df['std NIPS22'] )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:53:51.635343Z","iopub.execute_input":"2023-01-02T21:53:51.635774Z","iopub.status.idle":"2023-01-02T21:53:52.089709Z","shell.execute_reply.started":"2023-01-02T21:53:51.635739Z","shell.execute_reply":"2023-01-02T21:53:52.088775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x = df['Protein mean NIPS22'], y=df['Protein std NIPS22'])# *df['std NIPS22'] )","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:54:03.743299Z","iopub.execute_input":"2023-01-02T21:54:03.743782Z","iopub.status.idle":"2023-01-02T21:54:04.017970Z","shell.execute_reply.started":"2023-01-02T21:54:03.743740Z","shell.execute_reply":"2023-01-02T21:54:04.017031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.scatterplot(x = df['RNA mean NIPS22'], y=df['Protein mean NIPS22'])# *df['std NIPS22'] )","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:54:29.211508Z","iopub.execute_input":"2023-01-02T21:54:29.212687Z","iopub.status.idle":"2023-01-02T21:54:29.477211Z","shell.execute_reply.started":"2023-01-02T21:54:29.212636Z","shell.execute_reply":"2023-01-02T21:54:29.475774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[ ['RNA mean NIPS22', 'Protein mean NIPS22']  ].corr()","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:55:10.306122Z","iopub.execute_input":"2023-01-02T21:55:10.306998Z","iopub.status.idle":"2023-01-02T21:55:10.322930Z","shell.execute_reply.started":"2023-01-02T21:55:10.306946Z","shell.execute_reply":"2023-01-02T21:55:10.321387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[ ['RNA mean NIPS22', 'Protein mean NIPS22', 'RNA std NIPS22', 'Protein std NIPS22']  ].corr()","metadata":{"execution":{"iopub.status.busy":"2023-01-02T21:55:33.564124Z","iopub.execute_input":"2023-01-02T21:55:33.565290Z","iopub.status.idle":"2023-01-02T21:55:33.582226Z","shell.execute_reply.started":"2023-01-02T21:55:33.565225Z","shell.execute_reply":"2023-01-02T21:55:33.581070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:08:06.446076Z","iopub.execute_input":"2023-01-02T22:08:06.446488Z","iopub.status.idle":"2023-01-02T22:08:06.454013Z","shell.execute_reply.started":"2023-01-02T22:08:06.446454Z","shell.execute_reply":"2023-01-02T22:08:06.453031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor i in range(df.shape[0]): #  ,col in enumerate( df['CD_name']):# .columns:\n    IX = df.index[i]\n    cd = df[ 'CD_name' ].iat[i]\n    # print(cd, 'Secs passed: %.1f'%(time.time() - t0) ) \n    col = df[ 'ID in NIPS22 RNA dataset' ].iat[i]\n    if pd.isnull(col): continue \n\n    df.loc[IX,'Corr to RNA NIPS22'] = np.corrcoef(df_y[cd], df_rna[col])[0,1]\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:17:58.226861Z","iopub.execute_input":"2023-01-02T22:17:58.227328Z","iopub.status.idle":"2023-01-02T22:17:58.419281Z","shell.execute_reply.started":"2023-01-02T22:17:58.227290Z","shell.execute_reply":"2023-01-02T22:17:58.417773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport time\nt0 = time.time()\nv = np.zeros(df_rna.shape[1])\ns0 = pd.Series(index = df_rna.columns, data = v)\nfor i in range(df.shape[0]): #  ,col in enumerate( df['CD_name']):# .columns:\n    IX = df.index[i]\n    cd = df[ 'CD_name' ].iat[i]\n    print(cd, 'Secs passed: %.1f'%(time.time() - t0) ) \n    #col = df[ 'ID in NIPS22 RNA dataset' ].iat[i]\n    #if pd.isnull(col): continue \n    \n    for j,col in enumerate(df_rna.columns):\n        s0.iloc[j] = np.corrcoef(df_y[cd], df_rna[col])[0,1]\n    s = s0[s0.notnull()]\n    s = s.sort_values(ascending = False)\n    st = ''\n    for j in range(5):\n        g = s.index[j].split('_')[1]\n        c = np.round(s.iat[j],2)\n        st += g + ' (' + str(c) + ') ' \n    df.loc[IX,'Top5 Corr RNA NIPS22'] = st\n    st = ''\n    for j in [-1,-2,-3,-4,-5]:\n        g = s.index[j].split('_')[1]\n        c = np.round(s.iat[j],2)\n        st += g + ' (' + str(c) + ') ' \n    df.loc[IX,'Top5 AntiCorr RNA NIPS22'] = st\n    \n    \n    \ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:14:13.855678Z","iopub.execute_input":"2023-01-02T22:14:13.856183Z","iopub.status.idle":"2023-01-02T22:15:25.767134Z","shell.execute_reply.started":"2023-01-02T22:14:13.856141Z","shell.execute_reply":"2023-01-02T22:15:25.765904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:15:32.733455Z","iopub.execute_input":"2023-01-02T22:15:32.733905Z","iopub.status.idle":"2023-01-02T22:15:32.761017Z","shell.execute_reply.started":"2023-01-02T22:15:32.733865Z","shell.execute_reply":"2023-01-02T22:15:32.759838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add Information ","metadata":{}},{"cell_type":"code","source":"\n!pip install mygene\nimport mygene","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:18:48.352536Z","iopub.execute_input":"2023-01-02T22:18:48.353012Z","iopub.status.idle":"2023-01-02T22:19:03.768726Z","shell.execute_reply.started":"2023-01-02T22:18:48.352971Z","shell.execute_reply":"2023-01-02T22:19:03.767094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:20:43.861349Z","iopub.execute_input":"2023-01-02T22:20:43.862636Z","iopub.status.idle":"2023-01-02T22:20:43.871534Z","shell.execute_reply.started":"2023-01-02T22:20:43.862562Z","shell.execute_reply":"2023-01-02T22:20:43.869999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nlist_genes  = df['Ensembl ID'].tolist()\nprint(len(list_genes), list_genes[:5])\nmg = mygene.MyGeneInfo()\ng = mg.getgenes( list_genes,  fields=[ 'map_location', 'type_of_gene' ,   'summary' ], species='human'\n               , as_dataframe=True)# [:1000])\nprint(g.shape)\ng.head(50)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:23:45.301450Z","iopub.execute_input":"2023-01-02T22:23:45.302792Z","iopub.status.idle":"2023-01-02T22:23:47.106526Z","shell.execute_reply.started":"2023-01-02T22:23:45.302740Z","shell.execute_reply":"2023-01-02T22:23:47.105247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = g.reset_index()\ng.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:25:25.922256Z","iopub.execute_input":"2023-01-02T22:25:25.922728Z","iopub.status.idle":"2023-01-02T22:25:25.945649Z","shell.execute_reply.started":"2023-01-02T22:25:25.922688Z","shell.execute_reply":"2023-01-02T22:25:25.944712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = g.drop_duplicates('query')\nprint(g.shape)\nprint(g.columns)\ng.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:28:08.723068Z","iopub.execute_input":"2023-01-02T22:28:08.724235Z","iopub.status.idle":"2023-01-02T22:28:08.746824Z","shell.execute_reply.started":"2023-01-02T22:28:08.724183Z","shell.execute_reply":"2023-01-02T22:28:08.745660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:28:24.693928Z","iopub.execute_input":"2023-01-02T22:28:24.694486Z","iopub.status.idle":"2023-01-02T22:28:24.704124Z","shell.execute_reply.started":"2023-01-02T22:28:24.694432Z","shell.execute_reply":"2023-01-02T22:28:24.702828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = pd.merge(df, g[['query', 'map_location', 'type_of_gene', 'summary']] , left_on = 'Ensembl ID', right_on = 'query', how = 'left' )\nd = d.drop('query', axis = 1)\nprint(d.shape)\nprint(d.columns)\ndf = d\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:32:19.635087Z","iopub.execute_input":"2023-01-02T22:32:19.635611Z","iopub.status.idle":"2023-01-02T22:32:19.674503Z","shell.execute_reply.started":"2023-01-02T22:32:19.635574Z","shell.execute_reply":"2023-01-02T22:32:19.673550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:32:25.103804Z","iopub.execute_input":"2023-01-02T22:32:25.104510Z","iopub.status.idle":"2023-01-02T22:32:25.238815Z","shell.execute_reply.started":"2023-01-02T22:32:25.104469Z","shell.execute_reply":"2023-01-02T22:32:25.237435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = g['query'] == 'ENSG00000100031'\nprint( g[m]['summary'].values)","metadata":{"execution":{"iopub.status.busy":"2023-01-02T22:34:52.323584Z","iopub.execute_input":"2023-01-02T22:34:52.324034Z","iopub.status.idle":"2023-01-02T22:34:52.333853Z","shell.execute_reply.started":"2023-01-02T22:34:52.323998Z","shell.execute_reply":"2023-01-02T22:34:52.332467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('CD_information.csv')","metadata":{},"execution_count":null,"outputs":[]}]}