{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-06T19:35:14.003645Z","iopub.execute_input":"2022-11-06T19:35:14.004256Z","iopub.status.idle":"2022-11-06T19:35:14.026218Z","shell.execute_reply.started":"2022-11-06T19:35:14.004196Z","shell.execute_reply":"2022-11-06T19:35:14.025215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#I explore the gene expression data using scanpy on the CITE-seq data set and find \n#that markers CD3, CD4, CD8, CD56 and CD303 have higher gene expression levels on the training data set.","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:35:14.028427Z","iopub.execute_input":"2022-11-06T19:35:14.029225Z","iopub.status.idle":"2022-11-06T19:35:14.034024Z","shell.execute_reply.started":"2022-11-06T19:35:14.029180Z","shell.execute_reply":"2022-11-06T19:35:14.032510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n!pip install scanpy\n#!pip install scvelo\n#!pip install git+https://github.com/csgroen/scycle.git#egg=scycle\n#!pip install  --no-dependencies  git+https://github.com/j-bac/elpigraph-python.git\n# from elpigraph_ps_tools import *\n#import elpigraph\nimport anndata\n!pip install scanpy\nimport numpy as np\nimport pandas as pd\nimport scanpy as sc\nimport matplotlib.pyplot as plt\n\nsc.settings.verbosity = 3             # verbosity: errors (0), warnings (1), info (2), hints (3)\n#sc.logging.print_versions()\nimport importlib\n!pip install scanpy\nimport scanpy as sc\nimport matplotlib.pyplot as plt\n\nsc.settings.verbosity = 3             # verbosity: errors (0), warnings (1), info (2), hints (3)\n#sc.logging.print_versions()\n\nimport pandas as pd\nimport numpy as np\nfrom sklearn.decomposition import PCA\n\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nfrom sklearn.neighbors import NearestNeighbors\n\nmpl.rcParams['figure.dpi'] = 70\n\nimport os\n\nimport sys","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:35:14.035704Z","iopub.execute_input":"2022-11-06T19:35:14.036636Z","iopub.status.idle":"2022-11-06T19:35:36.766241Z","shell.execute_reply.started":"2022-11-06T19:35:14.036578Z","shell.execute_reply":"2022-11-06T19:35:36.764758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAIN_DIRECT = '../input/open-problems-multimodal'\nmeta_data_path = os.path.join(MAIN_DIRECT + '/metadata.csv')\neval_data_path  = os.path.join(MAIN_DIRECT + '/evaluation_ids.csv')\nsubmission_path = os.path.join(MAIN_DIRECT + '/sample_submission.csv')\n\ntest_cite_path = os.path.join(MAIN_DIRECT + '/test_cite_inputs.h5')\ntest_cite_path_day2 = os.path.join(MAIN_DIRECT + '/test_cite_inputs_day_2_donor_27678.h5')\ntest_multi_path = os.path.join(MAIN_DIRECT + '/test_multi_inputs.h5')\n\n\ntrain_cite_path = os.path.join(MAIN_DIRECT + '/train_cite_inputs.h5')\ntrain_cite_target_path = os.path.join(MAIN_DIRECT + '/train_cite_targets.h5')\ntrain_multi_path = os.path.join(MAIN_DIRECT + '/train_multi_inputs.h5')\ntrain_multi_target_path = os.path.join(MAIN_DIRECT + '/train_multi_targets.h5')","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:35:36.768776Z","iopub.execute_input":"2022-11-06T19:35:36.769294Z","iopub.status.idle":"2022-11-06T19:35:36.777625Z","shell.execute_reply.started":"2022-11-06T19:35:36.769242Z","shell.execute_reply":"2022-11-06T19:35:36.776535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import h5py    \nimport numpy as np ","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:35:36.781886Z","iopub.execute_input":"2022-11-06T19:35:36.782343Z","iopub.status.idle":"2022-11-06T19:35:36.788403Z","shell.execute_reply.started":"2022-11-06T19:35:36.782300Z","shell.execute_reply":"2022-11-06T19:35:36.787333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import h5py\nstore = h5py.File(train_cite_path,'r')","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:35:36.789969Z","iopub.execute_input":"2022-11-06T19:35:36.790569Z","iopub.status.idle":"2022-11-06T19:35:36.804260Z","shell.execute_reply.started":"2022-11-06T19:35:36.790527Z","shell.execute_reply":"2022-11-06T19:35:36.803311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib.colors import ListedColormap, LinearSegmentedColormap\nfrom matplotlib.offsetbox import AnnotationBbox, OffsetImage\nfrom sklearn.decomposition import TruncatedSVD\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport itertools\nimport warnings\nimport os\nimport gc\nimport scanpy as sc","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:47:58.348130Z","iopub.execute_input":"2022-11-06T19:47:58.348702Z","iopub.status.idle":"2022-11-06T19:47:58.663210Z","shell.execute_reply.started":"2022-11-06T19:47:58.348660Z","shell.execute_reply":"2022-11-06T19:47:58.661863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = store.get('train_cite_inputs')\ndataset = np.array(data)\nprint(dataset.shape)\nk = list(data.attrs.keys())\nv = list(data.attrs.values())\nprint(k)\nprint(v)\n\ndel data\ndel dataset\ndel k,v\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:35:36.831742Z","iopub.execute_input":"2022-11-06T19:35:36.832080Z","iopub.status.idle":"2022-11-06T19:35:37.479479Z","shell.execute_reply.started":"2022-11-06T19:35:36.832047Z","shell.execute_reply":"2022-11-06T19:35:37.478223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" !pip install anndata","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:53:55.713214Z","iopub.execute_input":"2022-11-06T19:53:55.714038Z","iopub.status.idle":"2022-11-06T19:54:06.955239Z","shell.execute_reply.started":"2022-11-06T19:53:55.713976Z","shell.execute_reply":"2022-11-06T19:54:06.953914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_hdf(train_cite_path)","metadata":{"execution":{"iopub.status.busy":"2022-11-06T20:15:07.329525Z","iopub.execute_input":"2022-11-06T20:15:07.330027Z","iopub.status.idle":"2022-11-06T20:16:09.265455Z","shell.execute_reply.started":"2022-11-06T20:15:07.329981Z","shell.execute_reply":"2022-11-06T20:16:09.264084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df,'Train HDF Data')","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:36:31.182961Z","iopub.execute_input":"2022-11-06T19:36:31.183406Z","iopub.status.idle":"2022-11-06T19:36:31.250679Z","shell.execute_reply.started":"2022-11-06T19:36:31.183364Z","shell.execute_reply":"2022-11-06T19:36:31.249538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata = sc.AnnData(df)","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:36:43.042860Z","iopub.execute_input":"2022-11-06T19:36:43.043279Z","iopub.status.idle":"2022-11-06T19:36:43.079393Z","shell.execute_reply.started":"2022-11-06T19:36:43.043240Z","shell.execute_reply":"2022-11-06T19:36:43.078318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(adata)","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:36:43.086792Z","iopub.execute_input":"2022-11-06T19:36:43.087156Z","iopub.status.idle":"2022-11-06T19:36:43.093919Z","shell.execute_reply.started":"2022-11-06T19:36:43.087112Z","shell.execute_reply":"2022-11-06T19:36:43.092505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport scanpy as sc","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:36:43.095453Z","iopub.execute_input":"2022-11-06T19:36:43.095793Z","iopub.status.idle":"2022-11-06T19:36:43.107179Z","shell.execute_reply.started":"2022-11-06T19:36:43.095759Z","shell.execute_reply":"2022-11-06T19:36:43.106184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.settings.verbosity = 3             # verbosity: errors (0), warnings (1), info (2), hints (3)\nsc.logging.print_header()\nsc.settings.set_figure_params(dpi=80, facecolor='white')","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:36:43.110262Z","iopub.execute_input":"2022-11-06T19:36:43.111319Z","iopub.status.idle":"2022-11-06T19:36:43.370591Z","shell.execute_reply.started":"2022-11-06T19:36:43.111283Z","shell.execute_reply":"2022-11-06T19:36:43.369233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata.var_names_make_unique() ","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:36:43.372193Z","iopub.execute_input":"2022-11-06T19:36:43.373241Z","iopub.status.idle":"2022-11-06T19:36:43.384478Z","shell.execute_reply.started":"2022-11-06T19:36:43.373195Z","shell.execute_reply":"2022-11-06T19:36:43.383212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:36:43.386244Z","iopub.execute_input":"2022-11-06T19:36:43.387502Z","iopub.status.idle":"2022-11-06T19:36:43.411077Z","shell.execute_reply.started":"2022-11-06T19:36:43.387455Z","shell.execute_reply":"2022-11-06T19:36:43.404919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pl.highest_expr_genes(adata, n_top=20, )","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:36:43.414530Z","iopub.execute_input":"2022-11-06T19:36:43.415707Z","iopub.status.idle":"2022-11-06T19:36:58.134304Z","shell.execute_reply.started":"2022-11-06T19:36:43.415655Z","shell.execute_reply":"2022-11-06T19:36:58.133152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scanpy\nimport scanpy as sc\nimport numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:44:32.198726Z","iopub.execute_input":"2022-11-06T19:44:32.199383Z","iopub.status.idle":"2022-11-06T19:44:43.513952Z","shell.execute_reply.started":"2022-11-06T19:44:32.199319Z","shell.execute_reply":"2022-11-06T19:44:43.512566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pp.filter_cells(adata, min_genes=200)\nsc.pp.filter_genes(adata, min_cells=3)","metadata":{"execution":{"iopub.status.busy":"2022-11-06T19:44:43.516843Z","iopub.execute_input":"2022-11-06T19:44:43.517396Z","iopub.status.idle":"2022-11-06T19:44:43.554918Z","shell.execute_reply.started":"2022-11-06T19:44:43.517340Z","shell.execute_reply":"2022-11-06T19:44:43.553233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata.var['mt'] = adata.var_names.str.startswith('MT-')  # annotate the group of mitochondrial genes as 'mt'\nsc.pp.calculate_qc_metrics(adata, qc_vars=['mt'], percent_top=None, log1p=False, inplace=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata = adata[adata.obs.n_genes_by_counts < 2500, :]\nadata = adata[adata.obs.pct_counts_mt < 5, :]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pp.normalize_total(adata, target_sum=1e4)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pp.log1p(adata)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pp.highly_variable_genes(adata, min_mean=0.0125, max_mean=3, min_disp=0.5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsc.pl.highly_variable_genes(adata)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata.raw = adata","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata = adata[:, adata.var.highly_variable]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pp.regress_out(adata, ['total_counts', 'pct_counts_mt'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pp.scale(adata, max_value=10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.tl.pca(adata, svd_solver='arpack')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pp.neighbors(adata, n_neighbors=10, n_pcs=40)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip3 install leidenalg\n!pip install louvain","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.tl.leiden(adata)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.tl.umap(adata)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.tl.paga(adata)\nsc.pl.paga(adata, plot=False)  # remove `plot=False` if you want to see the coarse-grained graph\nsc.tl.umap(adata, init_pos='paga')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.tl.leiden(adata, key_added='clusters', resolution=0.6)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pl.umap(adata, color='clusters', add_outline=True, legend_loc='on data',\n               legend_fontsize=12, legend_fontoutline=2,frameon=False,\n               title='clustering of cells', palette='Set1')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"marker_genes_dict = {\n    'B-cell': ['ENSG00000129673_AANAT'],\n    'DC': ['ENSG00000033050_ABCF2'],\n    'Monocytes': ['ENSG00000161204_ABCF3', 'ENSG00000205002_AARD'],\n    'NK': ['ENSG00000183044_ABAT', 'ENSG00000127837_AAMP'],\n    'CD8': ['ENSG00000087884_AAMDC'],\n    'CD4': ['ENSG00000275700_AATF'],\n    'T-cell': ['ENSG00000115977_AAK1'],\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata.raw.var_names","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pl.dotplot(adata, marker_genes_dict, 'clusters', dendrogram=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"marker_genes_dict2 = {\n    'CD38': ['ENSG00000185065_AC000068.1'],\n    'CD45RA': ['ENSG00000124608_AARS2'],\n    'CD62L': ['ENSG00000023839_ABCC2'],\n    'KLRG1': ['ENSG00000091262_ABCC6'],\n    'CD2': ['ENSG00000175164_ABO'],\n    'CD303': ['ENSG00000163322_ABRAXAS1'],\n    'CD163': ['ENSG00000097007_ABL1'],\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pl.dotplot(adata, marker_genes_dict2, 'clusters', dendrogram=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lineage specific markers\nmarker_genes_dict3 = {\n    'CD3': ['ENSG00000115977_AAK1'],\n    'CD8': ['ENSG00000087884_AAMDC'],\n    'CD56': ['ENSG00000127837_AAMP'],\n    'CD19': ['ENSG00000129673_AANAT'],\n    'CD4': ['ENSG00000275700_AATF'],\n    'CD14': ['ENSG00000281376_ABALON'],\n    'CD16': ['ENSG00000183044_ABAT'],\n    'CD11c': ['ENSG00000205002_AARD'],\n    'HLA-A-B-C': ['ENSG00000090861_AARS'],\n    'CD1c': ['ENSG00000033050_ABCF2'],\n    'CD11b': ['ENSG00000161204_ABCF3'],\n    'CD83': ['ENSG00000143322_ABL2'],\n    'CD73': ['ENSG00000265975_AC002091.1'],\n    'CD2': ['ENSG00000175164_ABO'],\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc.pl.dotplot(adata, marker_genes_dict3, 'clusters', dendrogram=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}