{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom scipy import stats\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-29T18:33:33.351124Z","iopub.execute_input":"2022-12-29T18:33:33.351617Z","iopub.status.idle":"2022-12-29T18:33:33.393547Z","shell.execute_reply.started":"2022-12-29T18:33:33.351524Z","shell.execute_reply":"2022-12-29T18:33:33.392437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count = pd.read_csv('../input/citeseqgse148127/GSE148127_ADT.counts.csv', index_col=0)\ncount.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T18:39:19.198061Z","iopub.execute_input":"2022-12-29T18:39:19.198633Z","iopub.status.idle":"2022-12-29T18:39:19.781439Z","shell.execute_reply.started":"2022-12-29T18:39:19.198589Z","shell.execute_reply":"2022-12-29T18:39:19.780555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RNA = pd.read_csv('../input/citeseqgse148127/GSE148127_SCT.normalized.RNA.counts.csv', index_col=0)\nRNA.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T18:41:20.831740Z","iopub.execute_input":"2022-12-29T18:41:20.832171Z","iopub.status.idle":"2022-12-29T18:42:17.127984Z","shell.execute_reply.started":"2022-12-29T18:41:20.832137Z","shell.execute_reply":"2022-12-29T18:42:17.126918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Source: https://www.kaggle.com/code/yoshifumimiya/simple-models-for-cds-all-proteins-gse148127/notebook?scriptVersionId=114395502\ntable = pd.read_csv('/kaggle/input/gse148127-gene-rna-list/GSE148127_gene_rna_list.csv')\ngene_id = table[table[\"In RNA dataset\"]=='Yes'][\"ID in RNA dataset\"]\nantibody_id = table[table[\"In RNA dataset\"]=='Yes'][\"Description\"]\ndict_list = dict(zip(gene_id, antibody_id))","metadata":{"execution":{"iopub.status.busy":"2022-12-29T18:48:16.224249Z","iopub.execute_input":"2022-12-29T18:48:16.224730Z","iopub.status.idle":"2022-12-29T18:48:16.236223Z","shell.execute_reply.started":"2022-12-29T18:48:16.224691Z","shell.execute_reply":"2022-12-29T18:48:16.234995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_list","metadata":{"execution":{"iopub.status.busy":"2022-12-29T18:48:18.958839Z","iopub.execute_input":"2022-12-29T18:48:18.959238Z","iopub.status.idle":"2022-12-29T18:48:18.968367Z","shell.execute_reply.started":"2022-12-29T18:48:18.959206Z","shell.execute_reply":"2022-12-29T18:48:18.967199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/dict-of-gse/CD_lasso_scores_sorted.csv', index_col=0)\ndf.reset_index(drop=True, inplace=True)\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:06:10.258712Z","iopub.execute_input":"2022-12-29T19:06:10.259132Z","iopub.status.idle":"2022-12-29T19:06:10.279261Z","shell.execute_reply.started":"2022-12-29T19:06:10.259101Z","shell.execute_reply":"2022-12-29T19:06:10.278078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Spearman correlation for each pair\nnames = []\ngenes = []\ncorrelations = []\nfor key, val in dict_list.items():\n    genes.append(key)\n    rna = RNA.loc[key,]\n    prot = count.loc[val,]\n    rho, pval = stats.spearmanr(rna, prot)\n    names.append(val)\n    correlations.append(rho)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:47:16.962120Z","iopub.execute_input":"2022-12-29T19:47:16.962745Z","iopub.status.idle":"2022-12-29T19:47:17.093091Z","shell.execute_reply.started":"2022-12-29T19:47:16.962701Z","shell.execute_reply":"2022-12-29T19:47:17.091838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = {'CD':names, 'Spearman correlation':correlations}\ndf_stat = pd.DataFrame(d)\nmerged_df = pd.merge(df_stat, df, on='CD')\nmerged_df","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:14:59.890139Z","iopub.execute_input":"2022-12-29T19:14:59.890570Z","iopub.status.idle":"2022-12-29T19:14:59.916108Z","shell.execute_reply.started":"2022-12-29T19:14:59.890537Z","shell.execute_reply":"2022-12-29T19:14:59.914624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged_df.to_csv('Spearman.csv')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:15:37.855550Z","iopub.execute_input":"2022-12-29T19:15:37.856012Z","iopub.status.idle":"2022-12-29T19:15:37.866108Z","shell.execute_reply.started":"2022-12-29T19:15:37.855975Z","shell.execute_reply":"2022-12-29T19:15:37.864678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plots","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:21:11.608348Z","iopub.execute_input":"2022-12-29T19:21:11.609835Z","iopub.status.idle":"2022-12-29T19:21:11.902973Z","shell.execute_reply.started":"2022-12-29T19:21:11.609782Z","shell.execute_reply":"2022-12-29T19:21:11.901630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full = count.index\nshort = list(merged_df['CD'])\nuse_col = []\nfor f in full:\n    if f in short:\n        use_col.append(f)\nprint(use_col)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:44:00.078171Z","iopub.execute_input":"2022-12-29T19:44:00.078691Z","iopub.status.idle":"2022-12-29T19:44:00.086271Z","shell.execute_reply.started":"2022-12-29T19:44:00.078650Z","shell.execute_reply":"2022-12-29T19:44:00.085055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prot_var = np.var(count.loc[use_col], axis=1)\nprot_mean = count.loc[use_col].mean(axis=1)\nrna_var = np.var(RNA.loc[genes], axis=1)\nrna_mean = RNA.loc[genes].mean(axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:47:39.984791Z","iopub.execute_input":"2022-12-29T19:47:39.985276Z","iopub.status.idle":"2022-12-29T19:47:40.142971Z","shell.execute_reply.started":"2022-12-29T19:47:39.985239Z","shell.execute_reply":"2022-12-29T19:47:40.141589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(prot_var))\nprint(len(rna_var))","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:47:44.324867Z","iopub.execute_input":"2022-12-29T19:47:44.325287Z","iopub.status.idle":"2022-12-29T19:47:44.333392Z","shell.execute_reply.started":"2022-12-29T19:47:44.325253Z","shell.execute_reply":"2022-12-29T19:47:44.331755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6), dpi=300)\nsns.scatterplot(x=prot_var, y=rna_var)\nplt.suptitle(\"Var(RNA) vs var(CD)\")\nplt.xlabel(\"var(CD)\")\nplt.ylabel(\"var(RNA)\")","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:51:38.898356Z","iopub.execute_input":"2022-12-29T19:51:38.899402Z","iopub.status.idle":"2022-12-29T19:51:39.398662Z","shell.execute_reply.started":"2022-12-29T19:51:38.899356Z","shell.execute_reply":"2022-12-29T19:51:39.397655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6), dpi=300)\nsns.scatterplot(x=prot_mean, y=prot_var)\nplt.suptitle(\"Var(CD) vs mean(CD)\")\nplt.xlabel(\"mean(CD)\")\nplt.ylabel(\"var(CD)\")","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:51:31.226045Z","iopub.execute_input":"2022-12-29T19:51:31.226556Z","iopub.status.idle":"2022-12-29T19:51:31.776794Z","shell.execute_reply.started":"2022-12-29T19:51:31.226509Z","shell.execute_reply":"2022-12-29T19:51:31.775668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6), dpi=300)\nsns.scatterplot(x=rna_mean, y=rna_var)\nplt.suptitle(\"Var(RNA) vs mean(RNA)\")\nplt.xlabel(\"mean(RNA)\")\nplt.ylabel(\"var(RNA)\")","metadata":{"execution":{"iopub.status.busy":"2022-12-29T19:51:17.681177Z","iopub.execute_input":"2022-12-29T19:51:17.681674Z","iopub.status.idle":"2022-12-29T19:51:18.225354Z","shell.execute_reply.started":"2022-12-29T19:51:17.681636Z","shell.execute_reply":"2022-12-29T19:51:18.224159Z"},"trusted":true},"execution_count":null,"outputs":[]}]}