{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-30T11:37:42.555024Z","iopub.execute_input":"2022-10-30T11:37:42.555529Z","iopub.status.idle":"2022-10-30T11:37:42.581231Z","shell.execute_reply.started":"2022-10-30T11:37:42.555439Z","shell.execute_reply":"2022-10-30T11:37:42.580239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nDATASET_DIR = \"/kaggle/input/cd-pathways-and-families/\"\nGENES_INFO = \"/kaggle/input/genes-information/\"\nSITESEQ_DENOISED = \"/kaggle/input/citeseq-denoised\"\n\nREACTOME_GENES = os.path.join(DATASET_DIR,\"gene_pathways.json\")\nGENE_GROUPS = os.path.join(DATASET_DIR,\"groups_data.json\")\nPATHWAY_DATA = os.path.join(DATASET_DIR, \"reactome_gene_info.tsv\")\nHGNC_DATA = os.path.join(GENES_INFO,\"hgnc_complete_set.txt\")\n\nMAGIC_TRAIN = os.path.join(SITESEQ_DENOISED, \"train_cite_inputs_denoised.h5\")\nMAGIC_TEST = os.path.join(SITESEQ_DENOISED, \"test_cite_inputs_denoised.h5\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:38:41.938420Z","iopub.execute_input":"2022-10-30T11:38:41.938834Z","iopub.status.idle":"2022-10-30T11:38:41.955201Z","shell.execute_reply.started":"2022-10-30T11:38:41.938800Z","shell.execute_reply":"2022-10-30T11:38:41.953609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pathways_names = pd.read_table(PATHWAY_DATA)\n# df_cite_train = pd.read_hdf(FP_CITE_TRAIN_INPUTS)\n# df_cite_test = pd.read_hdf(FP_CITE_TEST_INPUTS)\n\ndf_cite_train = pd.read_hdf(MAGIC_TRAIN)\ndf_cite_test = pd.read_hdf(MAGIC_TEST)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:38:48.112875Z","iopub.execute_input":"2022-10-30T11:38:48.113379Z","iopub.status.idle":"2022-10-30T11:41:14.480317Z","shell.execute_reply.started":"2022-10-30T11:38:48.113334Z","shell.execute_reply":"2022-10-30T11:41:14.478865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df_cite_train), len(df_cite_test)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:43:01.557535Z","iopub.execute_input":"2022-10-30T11:43:01.558046Z","iopub.status.idle":"2022-10-30T11:43:01.568941Z","shell.execute_reply.started":"2022-10-30T11:43:01.557959Z","shell.execute_reply":"2022-10-30T11:43:01.567502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_df = pd.concat([df_cite_train, df_cite_test])","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:43:28.671641Z","iopub.execute_input":"2022-10-30T11:43:28.672191Z","iopub.status.idle":"2022-10-30T11:43:35.642644Z","shell.execute_reply.started":"2022-10-30T11:43:28.672137Z","shell.execute_reply":"2022-10-30T11:43:35.641001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(full_df)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:43:35.645066Z","iopub.execute_input":"2022-10-30T11:43:35.645894Z","iopub.status.idle":"2022-10-30T11:43:35.654310Z","shell.execute_reply.started":"2022-10-30T11:43:35.645849Z","shell.execute_reply":"2022-10-30T11:43:35.652834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_cite_test\ndel df_cite_train","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:43:35.656396Z","iopub.execute_input":"2022-10-30T11:43:35.656910Z","iopub.status.idle":"2022-10-30T11:43:35.696325Z","shell.execute_reply.started":"2022-10-30T11:43:35.656863Z","shell.execute_reply":"2022-10-30T11:43:35.694680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(GENE_GROUPS, \"r\") as f:\n    pathways_data_json = json.load(f)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:43:53.316701Z","iopub.execute_input":"2022-10-30T11:43:53.317240Z","iopub.status.idle":"2022-10-30T11:43:53.342671Z","shell.execute_reply.started":"2022-10-30T11:43:53.317187Z","shell.execute_reply":"2022-10-30T11:43:53.341299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create map: gene name to ensembl id.\nhgnc_data = pd.read_table(HGNC_DATA)\nhgnc_data = hgnc_data[[\"ensembl_gene_id\", \"symbol\"]]\nhgnc_data_dict = dict(zip(hgnc_data[\"symbol\"], hgnc_data[\"ensembl_gene_id\"]))","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:44:31.066884Z","iopub.execute_input":"2022-10-30T11:44:31.067327Z","iopub.status.idle":"2022-10-30T11:44:32.020838Z","shell.execute_reply.started":"2022-10-30T11:44:31.067291Z","shell.execute_reply":"2022-10-30T11:44:32.019240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import defaultdict\npathways_full = defaultdict(set)\nfor key_gene, pathway_info in pathways_data_json.items():\n    for pathway_id, genes in pathway_info.items():\n        sample = {hgnc_data_dict.get(g) for g in genes}\n        if None in sample:\n            sample.remove(None)\n        pathways_full[pathway_id].update(sample)\n        pathways_full[pathway_id].add(hgnc_data_dict[key_gene])","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:44:34.849512Z","iopub.execute_input":"2022-10-30T11:44:34.849990Z","iopub.status.idle":"2022-10-30T11:44:34.872279Z","shell.execute_reply.started":"2022-10-30T11:44:34.849954Z","shell.execute_reply":"2022-10-30T11:44:34.870789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_names_map = {c.split(\"_\")[0]: c for c in full_df.columns}","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:44:36.779381Z","iopub.execute_input":"2022-10-30T11:44:36.779881Z","iopub.status.idle":"2022-10-30T11:44:36.798324Z","shell.execute_reply.started":"2022-10-30T11:44:36.779842Z","shell.execute_reply":"2022-10-30T11:44:36.797280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pathways_names.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:44:41.625461Z","iopub.execute_input":"2022-10-30T11:44:41.625947Z","iopub.status.idle":"2022-10-30T11:44:41.631796Z","shell.execute_reply.started":"2022-10-30T11:44:41.625910Z","shell.execute_reply":"2022-10-30T11:44:41.630256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# map_pathways_names = dict(zip(pathways_names[\"stId\"], pathways_names[\"displayName\"]))","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:44:42.119555Z","iopub.execute_input":"2022-10-30T11:44:42.120025Z","iopub.status.idle":"2022-10-30T11:44:42.125244Z","shell.execute_reply.started":"2022-10-30T11:44:42.119986Z","shell.execute_reply":"2022-10-30T11:44:42.123851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.mkdir(\"groups_info_magic\")","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:44:49.896755Z","iopub.execute_input":"2022-10-30T11:44:49.897224Z","iopub.status.idle":"2022-10-30T11:44:49.903469Z","shell.execute_reply.started":"2022-10-30T11:44:49.897181Z","shell.execute_reply":"2022-10-30T11:44:49.901844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = 0\nfor pathway_name, genes in pathways_full.items():\n    cur_cols = {col_names_map.get(g) for g in genes}\n    if None in cur_cols:\n        cur_cols.remove(None)\n        \n#     name = map_pathways_names[pathway_name] + \".csv\"\n#     name = name.replace(\"/\", \" \")\n#     full_df[list(cur_cols)].to_csv(os.path.join(\"/kaggle/working/pathway_info\", name), index=False)\n#     counter += 1\n#     print(counter)\n    \n    name = pathway_name + \".csv\"\n    name = name.replace(\" \", \"_\")\n    full_df[list(cur_cols)].to_csv(\n        os.path.join(\"/kaggle/working/groups_info_magic\", name), index=False\n    )\n    counter += 1\n    print(counter)","metadata":{"execution":{"iopub.status.busy":"2022-10-30T11:46:02.734649Z","iopub.execute_input":"2022-10-30T11:46:02.735153Z","iopub.status.idle":"2022-10-30T11:50:32.510435Z","shell.execute_reply.started":"2022-10-30T11:46:02.735117Z","shell.execute_reply":"2022-10-30T11:50:32.508900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}