{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Multimodal Single-Cell Integration Competition:  Tips for postprocessing \n\nSuggestion for Multiome prediction:\n\n1) \n\nDo normalization  to achieve :  (np.exp(Y)-1).sum(axis = 1) = 1e6 - because true targets satisfy that.  (Well, \"satisfy\" - up to machine precision, but nevertheless.   You can check - see below.) Why satisfy - that is standard biological processing - it is mentioned in data info tab, and well-known for everyone in the field. \n\nHow ? Can be different ways, but the most straighforward one:\n\nThat means:  1) calculate predictions Y 2) calculate normalizer Z =  sum(exp(Y)) 3) renorm:  Y_i -> Y_i + (log((1e6+22050 )/Z)) \n\n2) \n\nDo not fortget that true answers are non-negative - that is why it might be useful to clip negative predictions to zero - even before step 1. \n\n\n3) \n\nIt might be exp(Y) are INTs - but that is not clear for the moment - depends what soft organizers used. Will investigate later.  If it so: then use processing to achieve ints.  \n\n\nAnalysis as usually on Kaggle.\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T21:21:49.972077Z","iopub.execute_input":"2022-08-11T21:21:49.972474Z","iopub.status.idle":"2022-08-11T21:21:49.977181Z","shell.execute_reply.started":"2022-08-11T21:21:49.972443Z","shell.execute_reply":"2022-08-11T21:21:49.975948Z"}}},{"cell_type":"markdown","source":"# Install / import","metadata":{}},{"cell_type":"code","source":"#If you see a urllib warning running this cell, go to \"Settings\" on the right hand side, \n#and turn on internet. Note, you need to be phone verified.\n!pip install --quiet tables","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:23.484365Z","iopub.execute_input":"2022-08-31T10:27:23.484807Z","iopub.status.idle":"2022-08-31T10:27:34.597403Z","shell.execute_reply.started":"2022-08-31T10:27:23.484772Z","shell.execute_reply":"2022-08-31T10:27:34.596011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np \nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:34.599311Z","iopub.execute_input":"2022-08-31T10:27:34.599691Z","iopub.status.idle":"2022-08-31T10:27:34.606350Z","shell.execute_reply.started":"2022-08-31T10:27:34.599656Z","shell.execute_reply":"2022-08-31T10:27:34.605050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(\"/kaggle/input/open-problems-multimodal/\")","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:34.608013Z","iopub.execute_input":"2022-08-31T10:27:34.608903Z","iopub.status.idle":"2022-08-31T10:27:34.626445Z","shell.execute_reply.started":"2022-08-31T10:27:34.608855Z","shell.execute_reply":"2022-08-31T10:27:34.624793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare filenames and Load meta data","metadata":{}},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-31T10:27:34.629863Z","iopub.execute_input":"2022-08-31T10:27:34.630400Z","iopub.status.idle":"2022-08-31T10:27:34.639278Z","shell.execute_reply.started":"2022-08-31T10:27:34.630348Z","shell.execute_reply":"2022-08-31T10:27:34.637996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load scRNA-seq - train part of CITE-seq data","metadata":{}},{"cell_type":"code","source":"%%time\nSTART = 0; int(1e5)\nSTOP = START+1000 # 7e4\n\ndf_cite_train_x = pd.read_hdf(FP_CITE_TRAIN_INPUTS, start = START, stop = STOP )\ndisplay(df_cite_train_x)","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:34.640813Z","iopub.execute_input":"2022-08-31T10:27:34.641156Z","iopub.status.idle":"2022-08-31T10:27:35.652180Z","shell.execute_reply.started":"2022-08-31T10:27:34.641124Z","shell.execute_reply":"2022-08-31T10:27:35.650927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df_cite_train_x","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:35.653570Z","iopub.execute_input":"2022-08-31T10:27:35.653921Z","iopub.status.idle":"2022-08-31T10:27:35.659677Z","shell.execute_reply.started":"2022-08-31T10:27:35.653888Z","shell.execute_reply":"2022-08-31T10:27:35.658416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Check everything is non-negative:')\n(X < 0).sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:35.661564Z","iopub.execute_input":"2022-08-31T10:27:35.662159Z","iopub.status.idle":"2022-08-31T10:27:35.770231Z","shell.execute_reply.started":"2022-08-31T10:27:35.662113Z","shell.execute_reply":"2022-08-31T10:27:35.768912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Check normalization to 1e6. It seems more or less Okay, up to possibly machine precision: ')\n\nimport numpy as np \n(np.exp(X)-1).sum(axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:35.771865Z","iopub.execute_input":"2022-08-31T10:27:35.773021Z","iopub.status.idle":"2022-08-31T10:27:35.967709Z","shell.execute_reply.started":"2022-08-31T10:27:35.772967Z","shell.execute_reply":"2022-08-31T10:27:35.966376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Fast and dirty look - do we see INTs ? Seems no - but it might be machine precision - might no - investigate later: ')\nZ = (np.exp(X)-1)\nZ.iloc[:10,:20]","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:35.969026Z","iopub.execute_input":"2022-08-31T10:27:35.969374Z","iopub.status.idle":"2022-08-31T10:27:36.082936Z","shell.execute_reply.started":"2022-08-31T10:27:35.969343Z","shell.execute_reply":"2022-08-31T10:27:36.081816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load scRNA-seq data which are Multiome targets","metadata":{}},{"cell_type":"code","source":"%%time\nSTART = 0; int(1e5)\nSTOP = START+1000 # 7e4\ndf_multi_train_y = pd.read_hdf(FP_MULTIOME_TRAIN_TARGETS, start=START, stop=STOP) # Full file crashes kaggle memory \nprint(df_multi_train_y.shape)\ndisplay(df_multi_train_y.head() )","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:36.085959Z","iopub.execute_input":"2022-08-31T10:27:36.086376Z","iopub.status.idle":"2022-08-31T10:27:36.880607Z","shell.execute_reply.started":"2022-08-31T10:27:36.086333Z","shell.execute_reply":"2022-08-31T10:27:36.879419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Check everything is non-negative:')\n(X < 0).sum().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:36.881919Z","iopub.execute_input":"2022-08-31T10:27:36.882298Z","iopub.status.idle":"2022-08-31T10:27:36.984383Z","shell.execute_reply.started":"2022-08-31T10:27:36.882261Z","shell.execute_reply":"2022-08-31T10:27:36.983488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nprint('Check normalization to 1e6. It seems more or less Okay, up to possibly machine precision: ')\n\n(np.exp(X)-1).sum(axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:36.985703Z","iopub.execute_input":"2022-08-31T10:27:36.986465Z","iopub.status.idle":"2022-08-31T10:27:37.171301Z","shell.execute_reply.started":"2022-08-31T10:27:36.986425Z","shell.execute_reply":"2022-08-31T10:27:37.169986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Fast and dirty look - do we see INTs - seems no - but it might be machine precision - might no - investigate later: ')\nZ = (np.exp(X)-1)\nZ.iloc[:10,:20]","metadata":{"execution":{"iopub.status.busy":"2022-08-31T10:27:37.173287Z","iopub.execute_input":"2022-08-31T10:27:37.174156Z","iopub.status.idle":"2022-08-31T10:27:37.291014Z","shell.execute_reply.started":"2022-08-31T10:27:37.174117Z","shell.execute_reply":"2022-08-31T10:27:37.289763Z"},"trusted":true},"execution_count":null,"outputs":[]}]}