{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nGrandmaster Silogram kindly shared lists of his important features for the past competition on single cell data.\nhttps://www.kaggle.com/competitions/open-problems-multimodal/discussion/366455\n\nWe can see quite some biology from them - e.g. -  CD36 protein is higly activated in erythroid like cells - similar to red blood cells - so you can see genes like HBD, HBB, HBA1 as important features - that various forms  of the hemoglobin  - so quite as expected from biology.\n\nAlso look on the so-called genes enrichment analysis with KEGG pathways - again biologically reasonable results,\nand compared with importances by other methods - they are quite consistent.\n\nThat is a first look - we need some time to get more insights from the data. \n\n","metadata":{}},{"cell_type":"markdown","source":"# Load and look on data","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-15T17:17:50.698708Z","iopub.execute_input":"2022-12-15T17:17:50.699171Z","iopub.status.idle":"2022-12-15T17:17:50.720451Z","shell.execute_reply.started":"2022-12-15T17:17:50.699130Z","shell.execute_reply":"2022-12-15T17:17:50.719248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/pickle-convert/CSV_features_df_04_10CSV.csv'\ndf = pd.read_csv(fn)\ndf","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:17:50.722652Z","iopub.execute_input":"2022-12-15T17:17:50.723238Z","iopub.status.idle":"2022-12-15T17:17:50.839528Z","shell.execute_reply.started":"2022-12-15T17:17:50.723203Z","shell.execute_reply":"2022-12-15T17:17:50.838294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Look on feature importances for CD36 ","metadata":{}},{"cell_type":"code","source":"df[df['target'] == 'CD36']","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:17:50.840570Z","iopub.execute_input":"2022-12-15T17:17:50.840905Z","iopub.status.idle":"2022-12-15T17:17:50.870911Z","shell.execute_reply.started":"2022-12-15T17:17:50.840879Z","shell.execute_reply":"2022-12-15T17:17:50.869670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df[df['target'] == 'CD36'].to_csv('Silogram CD36 Feature importances.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:17:50.872521Z","iopub.execute_input":"2022-12-15T17:17:50.872918Z","iopub.status.idle":"2022-12-15T17:17:50.877862Z","shell.execute_reply.started":"2022-12-15T17:17:50.872880Z","shell.execute_reply":"2022-12-15T17:17:50.876562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfCD36 = df[df['target'] == 'CD36'].sort_values(by='Importance',ascending=False)\ndfCD36","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:17:50.880908Z","iopub.execute_input":"2022-12-15T17:17:50.881232Z","iopub.status.idle":"2022-12-15T17:17:50.913895Z","shell.execute_reply.started":"2022-12-15T17:17:50.881205Z","shell.execute_reply":"2022-12-15T17:17:50.913044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nv = dfCD36.iloc[:50]['Importance'].values\n#print(v)\nfig = plt.figure(figsize = (20,5))\nplt.plot(v , '*-' )\nplt.show()\nfig = plt.figure(figsize = (20,5))\nplt.bar( x= range(len(v)), height =  v)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:17:50.915192Z","iopub.execute_input":"2022-12-15T17:17:50.915544Z","iopub.status.idle":"2022-12-15T17:17:51.238499Z","shell.execute_reply.started":"2022-12-15T17:17:50.915503Z","shell.execute_reply":"2022-12-15T17:17:51.237245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nv = dfCD36.iloc[1:50]['Importance'].values\n#print(v)\nfig = plt.figure(figsize = (20,5))\nplt.plot(v , '*-' )\nplt.show()\nfig = plt.figure(figsize = (20,5))\nplt.bar( x= range(len(v)), height =  v)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:17:51.240503Z","iopub.execute_input":"2022-12-15T17:17:51.240970Z","iopub.status.idle":"2022-12-15T17:17:51.547951Z","shell.execute_reply.started":"2022-12-15T17:17:51.240927Z","shell.execute_reply":"2022-12-15T17:17:51.546573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nv = dfCD36.iloc[10:100]['Importance'].values\n#print(v)\nfig = plt.figure(figsize = (20,5))\nplt.plot(v , '*-' )\nplt.show()\nfig = plt.figure(figsize = (20,5))\nplt.bar( x= range(len(v)), height =  v)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:17:51.549303Z","iopub.execute_input":"2022-12-15T17:17:51.549697Z","iopub.status.idle":"2022-12-15T17:17:51.942320Z","shell.execute_reply.started":"2022-12-15T17:17:51.549664Z","shell.execute_reply":"2022-12-15T17:17:51.940969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nv = dfCD36.iloc[10:1000]['Importance'].values\n#print(v)\nv = np.log(1+v)\nfig = plt.figure(figsize = (20,5))\nplt.plot(v , '*-' )\nplt.show()\nfig = plt.figure(figsize = (20,5))\nplt.bar( x= range(len(v)), height =  v)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:17:51.943893Z","iopub.execute_input":"2022-12-15T17:17:51.944321Z","iopub.status.idle":"2022-12-15T17:17:53.845771Z","shell.execute_reply.started":"2022-12-15T17:17:51.944286Z","shell.execute_reply":"2022-12-15T17:17:53.844549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Genes sets enrichment analysis - what pathways are related to important genes ","metadata":{}},{"cell_type":"code","source":"!pip install gseapy","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:17:53.847155Z","iopub.execute_input":"2022-12-15T17:17:53.847511Z","iopub.status.idle":"2022-12-15T17:18:02.579348Z","shell.execute_reply.started":"2022-12-15T17:17:53.847483Z","shell.execute_reply":"2022-12-15T17:18:02.577783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gseapy as gp","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:18:02.581595Z","iopub.execute_input":"2022-12-15T17:18:02.582074Z","iopub.status.idle":"2022-12-15T17:18:02.587929Z","shell.execute_reply.started":"2022-12-15T17:18:02.582032Z","shell.execute_reply":"2022-12-15T17:18:02.586318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**А тут предполагалась сортировка?**","metadata":{}},{"cell_type":"code","source":"list_genes = df[df['target'] == 'CD36']['Feature'].tolist()\nlist_genes = list_genes[:100]\nlist_genes = [t.split('_')[1] for t in list_genes]\nprint(list_genes[:100])","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:18:02.589648Z","iopub.execute_input":"2022-12-15T17:18:02.590010Z","iopub.status.idle":"2022-12-15T17:18:02.617378Z","shell.execute_reply.started":"2022-12-15T17:18:02.589978Z","shell.execute_reply":"2022-12-15T17:18:02.615893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nenr = gp.enrichr(\n    gene_list=list_genes,\n    gene_sets=['KEGG_2016','KEGG_2021_Human'],\n    organism='human',\n    outdir=None,\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:18:02.618816Z","iopub.execute_input":"2022-12-15T17:18:02.619147Z","iopub.status.idle":"2022-12-15T17:18:06.715424Z","shell.execute_reply.started":"2022-12-15T17:18:02.619110Z","shell.execute_reply":"2022-12-15T17:18:06.714108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"enr.results.head(50)","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:18:06.718551Z","iopub.execute_input":"2022-12-15T17:18:06.718889Z","iopub.status.idle":"2022-12-15T17:18:06.746580Z","shell.execute_reply.started":"2022-12-15T17:18:06.718858Z","shell.execute_reply":"2022-12-15T17:18:06.744981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nlist_genes = df[df['target'] == 'CD36']['Feature'].tolist()\nlist_genes = list_genes[:500]\nlist_genes = [t.split('_')[1] for t in list_genes]\nprint(list_genes[:100])\n\nenr = gp.enrichr(\n    gene_list=list_genes,\n    gene_sets=['KEGG_2016','KEGG_2021_Human'],\n    organism='human',\n    outdir=None,\n)\n\nenr.results.head(50)","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:18:06.747898Z","iopub.execute_input":"2022-12-15T17:18:06.748874Z","iopub.status.idle":"2022-12-15T17:18:11.889684Z","shell.execute_reply.started":"2022-12-15T17:18:06.748820Z","shell.execute_reply":"2022-12-15T17:18:11.888837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compare with the other feature importances - correlations , catboost, etc..","metadata":{}},{"cell_type":"code","source":"fn = '/kaggle/input/research-project-01-around-multimodal-singlecell/CD36importances.csv'\ndf2 = pd.read_csv(fn)\ndf2","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:18:11.890921Z","iopub.execute_input":"2022-12-15T17:18:11.891378Z","iopub.status.idle":"2022-12-15T17:18:11.914803Z","shell.execute_reply.started":"2022-12-15T17:18:11.891343Z","shell.execute_reply":"2022-12-15T17:18:11.913348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2.columns = list( df2.iloc[0,:])\ndf2=df2.iloc[1:,:]\ndf2","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:18:11.916698Z","iopub.execute_input":"2022-12-15T17:18:11.917095Z","iopub.status.idle":"2022-12-15T17:18:11.939289Z","shell.execute_reply.started":"2022-12-15T17:18:11.917057Z","shell.execute_reply":"2022-12-15T17:18:11.938120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_stat = pd.DataFrame(); IX = 0\n\nfor N in [100,200,500,1000]:\n    list_genes = df[df['target'] == 'CD36']['Feature'].tolist()\n    list_genes = list_genes[:N]\n    \n\n    print('Interesection of Silogram  with others methods')\n    for col in df2.columns:\n        s = set([t.split('_')[1] for t in list_genes]) & set( [str(t).split('_')[-1] for t in df2[col]] )\n        #print(col, len(s))\n        df_stat.loc[col, 'Intersection with Silogram '+str(N)] = len(s)\n        df_stat.loc[col, 'Out of'] = len(set( df2[col]) )\n\ndf_stat    ","metadata":{"execution":{"iopub.status.busy":"2022-12-15T17:18:11.940860Z","iopub.execute_input":"2022-12-15T17:18:11.941647Z","iopub.status.idle":"2022-12-15T17:18:12.054919Z","shell.execute_reply.started":"2022-12-15T17:18:11.941609Z","shell.execute_reply":"2022-12-15T17:18:12.053404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}