{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nWe look on genes groups by name - e.g. many genes have names like CD1, CD2, CD3 ... or ZNF1,ZNF2, ZNF3 ...\nso the name indicates some similarity between their biological functions.\nBut some similarity might be fake - be careful. \n\n\nZNF - zinc finger family of proteins, LINC - some long non-coding rna which may be not well decribed, TMEM - transmembrane proteins, CD - cluster of differentiation, RPS - ribosomal S-proteins, RPL - ribosomal L-proteins \nEIF - Eukaryotic Initiation Factors Gene Family , HIST - histone genes, MIR - microRNA (mainly),\nZBTB - Zinc finger and BTB domain containing, CCDC - Coiled-coil domain-containing protein, \nGPR - G Protein-Coupled Receptor\n    \nSome gene families can be found here:\nhttps://en.wikipedia.org/wiki/List_of_gene_families , but not well correspond to groups below. \n\nThe top name prefixes what we get are:\n    \n    AC        3021\n    AL        1115\n    ZNF        538\n    LINC       367\n    AP         283\n    SLC        282\n    C          278\n    RPL        246\n    FAM        209\n    TMEM       208\n    RPS        153\n    RN         145\n    CCDC       118\n    ATP         97\n    RF          91\n    CD          83\n    EIF         78\n    RNF         77\n    PPP         74\n    RAB         70\n    HIST        68\n    RNU         64\n    WDR         62\n    LRRC        60\n    USP         59\n    ANKRD       56\n    MRPL        54\n    TRIM        53\n    UBE         51\n    IL          51\n    OR          50\n    Z           49\n    MAP         47\n    DDX         47\n    GPR         46\n    RBM         45\n    ZBTB        44\n    TTC         40\n    MIR         40\n    KIAA        40\n    KIF         39\n    CEP         39\n    SH          38\n    VPS         37\n    ARHGAP      36\n    CYP         34\n    MRPS        34\n    CDC         34\n    GTF         34\n    SNX         34","metadata":{}},{"cell_type":"markdown","source":"# Preparations","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-16T13:01:37.831506Z","iopub.execute_input":"2022-12-16T13:01:37.831930Z","iopub.status.idle":"2022-12-16T13:01:37.886940Z","shell.execute_reply.started":"2022-12-16T13:01:37.831895Z","shell.execute_reply":"2022-12-16T13:01:37.885757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --quiet tables","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:37.889193Z","iopub.execute_input":"2022-12-16T13:01:37.889619Z","iopub.status.idle":"2022-12-16T13:01:49.519993Z","shell.execute_reply.started":"2022-12-16T13:01:37.889583Z","shell.execute_reply":"2022-12-16T13:01:49.518358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:49.522102Z","iopub.execute_input":"2022-12-16T13:01:49.522525Z","iopub.status.idle":"2022-12-16T13:01:49.529938Z","shell.execute_reply.started":"2022-12-16T13:01:49.522488Z","shell.execute_reply":"2022-12-16T13:01:49.528370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:49.531790Z","iopub.execute_input":"2022-12-16T13:01:49.532214Z","iopub.status.idle":"2022-12-16T13:01:49.543985Z","shell.execute_reply.started":"2022-12-16T13:01:49.532176Z","shell.execute_reply":"2022-12-16T13:01:49.542732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_cite_train_x = pd.read_hdf(FP_CITE_TRAIN_INPUTS, stop=100)\n# df_cite_test_x = pd.read_hdf(FP_CITE_TEST_INPUTS)\ndf_cite_train_x.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:49.547114Z","iopub.execute_input":"2022-12-16T13:01:49.547571Z","iopub.status.idle":"2022-12-16T13:01:49.796723Z","shell.execute_reply.started":"2022-12-16T13:01:49.547535Z","shell.execute_reply":"2022-12-16T13:01:49.795531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df_cite_train_x","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:49.798449Z","iopub.execute_input":"2022-12-16T13:01:49.799247Z","iopub.status.idle":"2022-12-16T13:01:49.804153Z","shell.execute_reply.started":"2022-12-16T13:01:49.799209Z","shell.execute_reply":"2022-12-16T13:01:49.802766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Print main prefixes defining groups","metadata":{}},{"cell_type":"code","source":"%%time\nimport re\nlist_genes = [t.split('_')[-1] for t in df.columns ] \nprint(len(list_genes) , list_genes[:5] )\nlist_genes_prefixes = [re.sub(\"\\d\", \" \", s).split(' ')[0] for s in list_genes ]\nprint(len(list_genes_prefixes) , list_genes_prefixes[:50] )\npd.Series(list_genes_prefixes).value_counts().head(50)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:49.805707Z","iopub.execute_input":"2022-12-16T13:01:49.806298Z","iopub.status.idle":"2022-12-16T13:01:49.884353Z","shell.execute_reply.started":"2022-12-16T13:01:49.806237Z","shell.execute_reply":"2022-12-16T13:01:49.882836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Some statistics on groups ","metadata":{}},{"cell_type":"code","source":"pd.Series(list_genes_prefixes).value_counts().describe()\n","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:49.886327Z","iopub.execute_input":"2022-12-16T13:01:49.886737Z","iopub.status.idle":"2022-12-16T13:01:49.907700Z","shell.execute_reply.started":"2022-12-16T13:01:49.886700Z","shell.execute_reply":"2022-12-16T13:01:49.906195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (20,5))\nplt.plot(pd.Series(list_genes_prefixes).value_counts().head(30), '*-')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:49.909214Z","iopub.execute_input":"2022-12-16T13:01:49.909643Z","iopub.status.idle":"2022-12-16T13:01:50.236281Z","shell.execute_reply.started":"2022-12-16T13:01:49.909608Z","shell.execute_reply":"2022-12-16T13:01:50.234986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (20,5))\nplt.hist(pd.Series(list_genes_prefixes).value_counts(), bins = 100)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:50.238347Z","iopub.execute_input":"2022-12-16T13:01:50.238842Z","iopub.status.idle":"2022-12-16T13:01:50.637513Z","shell.execute_reply.started":"2022-12-16T13:01:50.238794Z","shell.execute_reply":"2022-12-16T13:01:50.636167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize = (20,5))\nplt.plot(np.log(1+ pd.Series(list_genes_prefixes).value_counts().values), '*-')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:50.639345Z","iopub.execute_input":"2022-12-16T13:01:50.639859Z","iopub.status.idle":"2022-12-16T13:01:50.891014Z","shell.execute_reply.started":"2022-12-16T13:01:50.639815Z","shell.execute_reply":"2022-12-16T13:01:50.889517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for grp in ['MIR','CCDC','GPR']:\n    print(); print(grp)\n    for t in df.columns:\n        if t.split('_')[-1].startswith(grp):\n            print(t)","metadata":{"execution":{"iopub.status.busy":"2022-12-16T13:01:50.893200Z","iopub.execute_input":"2022-12-16T13:01:50.893852Z","iopub.status.idle":"2022-12-16T13:01:50.943538Z","shell.execute_reply.started":"2022-12-16T13:01:50.893798Z","shell.execute_reply":"2022-12-16T13:01:50.941907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}