{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-16T18:39:58.027772Z","iopub.execute_input":"2022-10-16T18:39:58.028133Z","iopub.status.idle":"2022-10-16T18:39:58.035860Z","shell.execute_reply.started":"2022-10-16T18:39:58.028102Z","shell.execute_reply":"2022-10-16T18:39:58.034953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#the excellent notebook of usamabalochhh helped me to understand the project better and I tried new \n#visuals with histograms and other plots to explore the data better!\n#https://www.kaggle.com/code/usamabalochhh/single-cell-integration-explore-data","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib.colors import ListedColormap, LinearSegmentedColormap\nfrom matplotlib.offsetbox import AnnotationBbox, OffsetImage\nfrom sklearn.decomposition import TruncatedSVD\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport itertools\nimport warnings\nimport os\nimport gc","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:39:58.037613Z","iopub.execute_input":"2022-10-16T18:39:58.037914Z","iopub.status.idle":"2022-10-16T18:39:59.020873Z","shell.execute_reply.started":"2022-10-16T18:39:58.037888Z","shell.execute_reply":"2022-10-16T18:39:59.019771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#the following path joins are from the excellent notebook of usamabalochhh, \n#https://www.kaggle.com/code/usamabalochhh/single-cell-integration-explore-data","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:39:59.022443Z","iopub.execute_input":"2022-10-16T18:39:59.022744Z","iopub.status.idle":"2022-10-16T18:39:59.025973Z","shell.execute_reply.started":"2022-10-16T18:39:59.022715Z","shell.execute_reply":"2022-10-16T18:39:59.025212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAIN_DIRECT = '../input/open-problems-multimodal'\nmeta_data_path = os.path.join(MAIN_DIRECT + '/metadata.csv')\neval_data_path  = os.path.join(MAIN_DIRECT + '/evaluation_ids.csv')\nsubmission_path = os.path.join(MAIN_DIRECT + '/sample_submission.csv')\n\ntest_cite_path = os.path.join(MAIN_DIRECT + '/test_cite_inputs.h5')\ntest_cite_path_day2 = os.path.join(MAIN_DIRECT + '/test_cite_inputs_day_2_donor_27678.h5')\ntest_multi_path = os.path.join(MAIN_DIRECT + '/test_multi_inputs.h5')\n\n\ntrain_cite_path = os.path.join(MAIN_DIRECT + '/train_cite_inputs.h5')\ntrain_cite_target_path = os.path.join(MAIN_DIRECT + '/train_cite_targets.h5')\ntrain_multi_path = os.path.join(MAIN_DIRECT + '/train_multi_inputs.h5')\ntrain_multi_target_path = os.path.join(MAIN_DIRECT + '/train_multi_targets.h5')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:39:59.027964Z","iopub.execute_input":"2022-10-16T18:39:59.028895Z","iopub.status.idle":"2022-10-16T18:39:59.037257Z","shell.execute_reply.started":"2022-10-16T18:39:59.028863Z","shell.execute_reply":"2022-10-16T18:39:59.036467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data = pd.read_csv(meta_data_path)\ndisplay(meta_data,'Meta Data')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:39:59.039323Z","iopub.execute_input":"2022-10-16T18:39:59.040081Z","iopub.status.idle":"2022-10-16T18:39:59.383834Z","shell.execute_reply.started":"2022-10-16T18:39:59.040055Z","shell.execute_reply":"2022-10-16T18:39:59.383203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH_DATASET = \"/kaggle/input/open-problems-multimodal\"\ndf_meta = pd.read_csv(os.path.join(PATH_DATASET, \"metadata.csv\"))\ndisplay(df_meta.head())","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:39:59.384916Z","iopub.execute_input":"2022-10-16T18:39:59.385738Z","iopub.status.idle":"2022-10-16T18:39:59.598141Z","shell.execute_reply.started":"2022-10-16T18:39:59.385708Z","shell.execute_reply":"2022-10-16T18:39:59.597139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"citeseq_data = meta_data[meta_data['technology'] == 'citeseq']\ndisplay(citeseq_data,'CITEseq Data')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:39:59.599461Z","iopub.execute_input":"2022-10-16T18:39:59.600546Z","iopub.status.idle":"2022-10-16T18:39:59.641968Z","shell.execute_reply.started":"2022-10-16T18:39:59.600512Z","shell.execute_reply":"2022-10-16T18:39:59.641283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"citeseq_data.hist(column='day',by='cell_type')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:39:59.643066Z","iopub.execute_input":"2022-10-16T18:39:59.643359Z","iopub.status.idle":"2022-10-16T18:40:00.390275Z","shell.execute_reply.started":"2022-10-16T18:39:59.643332Z","shell.execute_reply":"2022-10-16T18:40:00.389372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"citeseq_data['cell_type'].hist()","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:00.391278Z","iopub.execute_input":"2022-10-16T18:40:00.391539Z","iopub.status.idle":"2022-10-16T18:40:00.613442Z","shell.execute_reply.started":"2022-10-16T18:40:00.391505Z","shell.execute_reply":"2022-10-16T18:40:00.612613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n%matplotlib inline\nciteseq_data.hist(column='cell_type',by='day',layout=(2,2),figsize=(8,8),sharex=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:00.616202Z","iopub.execute_input":"2022-10-16T18:40:00.616542Z","iopub.status.idle":"2022-10-16T18:40:01.189910Z","shell.execute_reply.started":"2022-10-16T18:40:00.616516Z","shell.execute_reply":"2022-10-16T18:40:01.188986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"citeseq_data.plot.scatter(x='day', y='cell_type', figsize=[4,4])","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:01.191075Z","iopub.execute_input":"2022-10-16T18:40:01.191385Z","iopub.status.idle":"2022-10-16T18:40:01.721865Z","shell.execute_reply.started":"2022-10-16T18:40:01.191358Z","shell.execute_reply":"2022-10-16T18:40:01.720955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"citeseq_data.hist(column='cell_type',by='donor',layout=(2,2),figsize=(8,8),sharex=True)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:01.723055Z","iopub.execute_input":"2022-10-16T18:40:01.723333Z","iopub.status.idle":"2022-10-16T18:40:02.377741Z","shell.execute_reply.started":"2022-10-16T18:40:01.723307Z","shell.execute_reply":"2022-10-16T18:40:02.376738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lmplot(x='day', y='donor',data=citeseq_data,hue='cell_type')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:02.378711Z","iopub.execute_input":"2022-10-16T18:40:02.378973Z","iopub.status.idle":"2022-10-16T18:40:10.344126Z","shell.execute_reply.started":"2022-10-16T18:40:02.378949Z","shell.execute_reply":"2022-10-16T18:40:10.343144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import h5py\nstore = h5py.File(train_cite_path,'r')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:10.345546Z","iopub.execute_input":"2022-10-16T18:40:10.346218Z","iopub.status.idle":"2022-10-16T18:40:10.493058Z","shell.execute_reply.started":"2022-10-16T18:40:10.346178Z","shell.execute_reply":"2022-10-16T18:40:10.492169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#the following hdf readings are from the excellent notebook of usamabalochhh, \n#https://www.kaggle.com/code/usamabalochhh/single-cell-integration-explore-data","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = store.get('train_cite_inputs')\ndataset = np.array(data)\nprint(dataset.shape)\nk = list(data.attrs.keys())\nv = list(data.attrs.values())\nprint(k)\nprint(v)\n\ndel data\ndel dataset\ndel k,v\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:10.494382Z","iopub.execute_input":"2022-10-16T18:40:10.495036Z","iopub.status.idle":"2022-10-16T18:40:10.627673Z","shell.execute_reply.started":"2022-10-16T18:40:10.494992Z","shell.execute_reply":"2022-10-16T18:40:10.626861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_hdf_cite = pd.read_hdf(train_cite_path)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:10.628698Z","iopub.execute_input":"2022-10-16T18:40:10.628974Z","iopub.status.idle":"2022-10-16T18:40:57.870724Z","shell.execute_reply.started":"2022-10-16T18:40:10.628948Z","shell.execute_reply":"2022-10-16T18:40:57.869449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_hdf_cite.info(memory_usage='deep')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:57.872213Z","iopub.execute_input":"2022-10-16T18:40:57.872543Z","iopub.status.idle":"2022-10-16T18:40:59.077941Z","shell.execute_reply.started":"2022-10-16T18:40:57.872512Z","shell.execute_reply":"2022-10-16T18:40:59.076938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_hdf_cite_target = pd.read_hdf(train_cite_target_path)\ndisplay(train_hdf_cite_target,'Train HDF Target Data')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:59.079530Z","iopub.execute_input":"2022-10-16T18:40:59.080153Z","iopub.status.idle":"2022-10-16T18:40:59.701136Z","shell.execute_reply.started":"2022-10-16T18:40:59.080111Z","shell.execute_reply":"2022-10-16T18:40:59.700231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_columns = train_hdf_cite.columns\ntrain_cite_target_columns = train_hdf_cite_target.columns\ntrain_cite_target_columns\ndf = pd.DataFrame()\ni = 0 \nfor (gene,prot) in zip(train_cite_columns,train_cite_target_columns):\n    df.at[i,'Gene'] = gene\n    df.at[i,'Protein'] = prot\n    i = i + 1","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:59.702636Z","iopub.execute_input":"2022-10-16T18:40:59.703112Z","iopub.status.idle":"2022-10-16T18:40:59.781850Z","shell.execute_reply.started":"2022-10-16T18:40:59.703072Z","shell.execute_reply":"2022-10-16T18:40:59.781197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df.head())","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:59.782780Z","iopub.execute_input":"2022-10-16T18:40:59.783101Z","iopub.status.idle":"2022-10-16T18:40:59.790888Z","shell.execute_reply.started":"2022-10-16T18:40:59.783077Z","shell.execute_reply":"2022-10-16T18:40:59.790060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf6 = pd.merge(train_hdf_cite, train_hdf_cite_target, on='cell_id')","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:40:59.792567Z","iopub.execute_input":"2022-10-16T18:40:59.792993Z","iopub.status.idle":"2022-10-16T18:41:05.246199Z","shell.execute_reply.started":"2022-10-16T18:40:59.792955Z","shell.execute_reply":"2022-10-16T18:41:05.245434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df6)","metadata":{"execution":{"iopub.status.busy":"2022-10-16T18:41:05.247370Z","iopub.execute_input":"2022-10-16T18:41:05.247819Z","iopub.status.idle":"2022-10-16T18:41:05.298513Z","shell.execute_reply.started":"2022-10-16T18:41:05.247790Z","shell.execute_reply":"2022-10-16T18:41:05.297584Z"},"trusted":true},"execution_count":null,"outputs":[]}]}