{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ? \n\nAnalyse genes which are related to gender dependences.\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-15T16:57:52.835999Z","iopub.execute_input":"2023-03-15T16:57:52.836427Z","iopub.status.idle":"2023-03-15T16:57:52.843343Z","shell.execute_reply.started":"2023-03-15T16:57:52.836387Z","shell.execute_reply":"2023-03-15T16:57:52.841997Z"}}},{"cell_type":"markdown","source":"# Preparations","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport time \nt0start = time.time()\nimport gc\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if 'research-project-01-around-multimodal-singlecell' not in os.path.join(dirname, filename):\n            print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-28T10:07:07.214621Z","iopub.execute_input":"2023-03-28T10:07:07.215548Z","iopub.status.idle":"2023-03-28T10:07:08.365092Z","shell.execute_reply.started":"2023-03-28T10:07:07.215502Z","shell.execute_reply":"2023-03-28T10:07:08.363655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Load data","metadata":{}},{"cell_type":"code","source":"%%time\n\n# Here we use more general setup which allows to with many datasets - from the Notebook: https://www.kaggle.com/code/alexandervc/many-citeseq-02-just-load-full-data\n# Presently we consider only one dataset so it is overkill, but nevertheless \n\ndict_df_prot = {}\ndict_df_rna = {}\ndict_df_meta = {}\n\nN_rows2take = int( 1e5 ) # Work only with first rows of the data - to speed up, avoid RAM crashes , etc... \nwork_with_rna_data = 'Yes'# If No - we will not store/work with protein data - only rna -  For some fast analysis we need only protein data, which much more smaller \nwork_with_protein_data = 'Yes' \ndict_datasets2consider = {}\ndict_datasets2consider['NIPS2022'] = 'Yes'\n\nimport gc\n\nif dict_datasets2consider['NIPS2022'] == 'Yes':\n\n    key4dict = 'NIPS2022'\n    \n    DATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\n    if work_with_protein_data == 'Yes':\n        FP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\n        df = pd.read_hdf(FP_CITE_TRAIN_TARGETS, start=0, stop=N_rows2take, )\n        print( [t for t in df.columns if 'CD45' in t.upper() ], [t for t in df.columns if 'CD53' in t.upper() ],  )\n        dict_df_prot[key4dict] = df\n        print('Protein data shape:', df.shape)\n        display(df.head(2)) \n\n    # %%time\n    if work_with_rna_data == 'Yes':\n        filename_rna_data = '/kaggle/input/open-problems-multimodal/train_cite_inputs.h5'\n        df = pd.read_hdf(filename_rna_data, start=0, stop=N_rows2take,)\n        l = [t.split('_')[1] for t in df.columns]; df.columns = l \n        index_save = df.index\n        print('RNA data shape:', df.shape)\n        dict_df_rna[key4dict] = df\n        display(df.head(2)) \n\n\n    fn = '/kaggle/input/open-problems-multimodal/metadata.csv'\n    df = pd.read_csv(fn, index_col = 0 );\n    mask = df.index.isin(index_save); df = df[mask]\n    print('Meta data shape:', df.shape)\n    dict_df_meta[key4dict] = df\n    display(df.head(2)) \n\n    gc.collect() ","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:07:08.367577Z","iopub.execute_input":"2023-03-28T10:07:08.368881Z","iopub.status.idle":"2023-03-28T10:08:15.939452Z","shell.execute_reply.started":"2023-03-28T10:07:08.368829Z","shell.execute_reply":"2023-03-28T10:08:15.937977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k in dict_df_meta:\n    df_prot = dict_df_prot[k]\n    df_rna = dict_df_rna[k]\n    df_meta = dict_df_meta[k]\n    print(df_prot.shape, df_rna.shape, df_meta.shape )","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:08:15.940944Z","iopub.execute_input":"2023-03-28T10:08:15.941289Z","iopub.status.idle":"2023-03-28T10:08:15.948704Z","shell.execute_reply.started":"2023-03-28T10:08:15.941255Z","shell.execute_reply":"2023-03-28T10:08:15.947322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta['donor'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:08:15.952348Z","iopub.execute_input":"2023-03-28T10:08:15.952706Z","iopub.status.idle":"2023-03-28T10:08:15.975192Z","shell.execute_reply.started":"2023-03-28T10:08:15.952673Z","shell.execute_reply":"2023-03-28T10:08:15.973826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n# Define gender variable: 1 - female, 0 - male","metadata":{}},{"cell_type":"code","source":"y = df_meta['donor'].isin([13176 ]).astype(float)\ny.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:08:15.976414Z","iopub.execute_input":"2023-03-28T10:08:15.977020Z","iopub.status.idle":"2023-03-28T10:08:15.993780Z","shell.execute_reply.started":"2023-03-28T10:08:15.976980Z","shell.execute_reply":"2023-03-28T10:08:15.992350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# roc_auc for cd-protein","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\nroc_auc_cd=pd.DataFrame(columns=df_prot.columns)\nfor i in df_prot.columns:\n    roc_auc_cd.loc[1, i]=roc_auc_score(y, df_prot[i])\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:08:15.995241Z","iopub.execute_input":"2023-03-28T10:08:15.996473Z","iopub.status.idle":"2023-03-28T10:08:19.389988Z","shell.execute_reply.started":"2023-03-28T10:08:15.996422Z","shell.execute_reply":"2023-03-28T10:08:19.388571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_cd=roc_auc_cd.T\nroc_auc_cd=roc_auc_cd.astype(float)\nroc_auc_cd=roc_auc_cd.sort_values(by=1, axis=0, ascending=False)\nroc_auc_cd.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:08:19.393981Z","iopub.execute_input":"2023-03-28T10:08:19.394824Z","iopub.status.idle":"2023-03-28T10:08:19.408660Z","shell.execute_reply.started":"2023-03-28T10:08:19.394784Z","shell.execute_reply":"2023-03-28T10:08:19.407160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# roc_auc for rna","metadata":{}},{"cell_type":"code","source":"print(df_rna.shape)\ndf_rna=df_rna.T\ndf_rna=df_rna[~df_rna.index.duplicated(keep='first')]\ndf_rna=df_rna.T\n\ndf_rna.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:08:19.410085Z","iopub.execute_input":"2023-03-28T10:08:19.410529Z","iopub.status.idle":"2023-03-28T10:08:21.513738Z","shell.execute_reply.started":"2023-03-28T10:08:19.410491Z","shell.execute_reply":"2023-03-28T10:08:21.512403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score\nroc_auc_rna=pd.DataFrame(columns=df_rna.columns)\nfor i in df_rna.columns:\n    roc_auc_rna.loc[1, i]=roc_auc_score(y, df_rna[i])","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:08:21.515414Z","iopub.execute_input":"2023-03-28T10:08:21.515938Z","iopub.status.idle":"2023-03-28T10:14:14.778250Z","shell.execute_reply.started":"2023-03-28T10:08:21.515872Z","shell.execute_reply":"2023-03-28T10:14:14.776887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_rna=roc_auc_rna.T\nroc_auc_rna=roc_auc_rna.astype(float)\nroc_auc_rna=roc_auc_rna.sort_values(by=1, axis=0, ascending=False)\nroc_auc_rna.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:14.782325Z","iopub.execute_input":"2023-03-28T10:14:14.782712Z","iopub.status.idle":"2023-03-28T10:14:14.803094Z","shell.execute_reply.started":"2023-03-28T10:14:14.782673Z","shell.execute_reply":"2023-03-28T10:14:14.802189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_cd.to_csv('roc_auc_cd.csv')\nroc_auc_rna.to_csv('roc_auc_rna.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:14.804305Z","iopub.execute_input":"2023-03-28T10:14:14.804612Z","iopub.status.idle":"2023-03-28T10:14:14.857549Z","shell.execute_reply.started":"2023-03-28T10:14:14.804581Z","shell.execute_reply":"2023-03-28T10:14:14.856656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# result","metadata":{}},{"cell_type":"markdown","source":"CD","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_columns', 50)\nroc_auc_cd.head(50)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:14.858815Z","iopub.execute_input":"2023-03-28T10:14:14.859353Z","iopub.status.idle":"2023-03-28T10:14:14.871583Z","shell.execute_reply.started":"2023-03-28T10:14:14.859319Z","shell.execute_reply":"2023-03-28T10:14:14.870483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_cd.tail(50)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:14.872928Z","iopub.execute_input":"2023-03-28T10:14:14.873210Z","iopub.status.idle":"2023-03-28T10:14:14.891684Z","shell.execute_reply.started":"2023-03-28T10:14:14.873181Z","shell.execute_reply":"2023-03-28T10:14:14.890375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_cd.describe()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:14.893703Z","iopub.execute_input":"2023-03-28T10:14:14.894178Z","iopub.status.idle":"2023-03-28T10:14:14.913505Z","shell.execute_reply.started":"2023-03-28T10:14:14.894130Z","shell.execute_reply":"2023-03-28T10:14:14.912063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(roc_auc_cd, bins=30, density=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:14.915497Z","iopub.execute_input":"2023-03-28T10:14:14.915967Z","iopub.status.idle":"2023-03-28T10:14:15.230066Z","shell.execute_reply.started":"2023-03-28T10:14:14.915919Z","shell.execute_reply":"2023-03-28T10:14:15.228959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"RNA\n","metadata":{}},{"cell_type":"code","source":"roc_auc_rna.head(50)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:15.231454Z","iopub.execute_input":"2023-03-28T10:14:15.231787Z","iopub.status.idle":"2023-03-28T10:14:15.245339Z","shell.execute_reply.started":"2023-03-28T10:14:15.231755Z","shell.execute_reply":"2023-03-28T10:14:15.243983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_rna.tail(50)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:15.247073Z","iopub.execute_input":"2023-03-28T10:14:15.247793Z","iopub.status.idle":"2023-03-28T10:14:15.264080Z","shell.execute_reply.started":"2023-03-28T10:14:15.247744Z","shell.execute_reply":"2023-03-28T10:14:15.262850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_rna.describe()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:15.265443Z","iopub.execute_input":"2023-03-28T10:14:15.265738Z","iopub.status.idle":"2023-03-28T10:14:15.283394Z","shell.execute_reply.started":"2023-03-28T10:14:15.265709Z","shell.execute_reply":"2023-03-28T10:14:15.282556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(roc_auc_rna, bins=30)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:15.284347Z","iopub.execute_input":"2023-03-28T10:14:15.284653Z","iopub.status.idle":"2023-03-28T10:14:15.568837Z","shell.execute_reply.started":"2023-03-28T10:14:15.284622Z","shell.execute_reply":"2023-03-28T10:14:15.567440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Trying to p_value","metadata":{}},{"cell_type":"code","source":"for i in roc_auc_rna.head(10).iterrows():\n    print(i)#sns.boxplot(x=)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:14:15.570501Z","iopub.execute_input":"2023-03-28T10:14:15.571411Z","iopub.status.idle":"2023-03-28T10:14:15.583807Z","shell.execute_reply.started":"2023-03-28T10:14:15.571360Z","shell.execute_reply":"2023-03-28T10:14:15.582641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_rna","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:48:06.907961Z","iopub.execute_input":"2023-03-28T10:48:06.909122Z","iopub.status.idle":"2023-03-28T10:48:07.041359Z","shell.execute_reply.started":"2023-03-28T10:48:06.909072Z","shell.execute_reply":"2023-03-28T10:48:07.039970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for j in df_meta['donor'].unique():\nfor i in df_rna.loc[df_rna.isin(roc_auc_rna.head(10))].iterrows():\n    sns.boxplot(x=df_meta['donor'], y=df_rna[i])","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:49:50.060199Z","iopub.execute_input":"2023-03-28T10:49:50.060627Z","iopub.status.idle":"2023-03-28T10:50:03.230363Z","shell.execute_reply.started":"2023-03-28T10:49:50.060587Z","shell.execute_reply":"2023-03-28T10:50:03.228385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}