{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Multimodal Single-Cell🧬IIntegration: EDA 🔍 & simple predictions","metadata":{}},{"cell_type":"code","source":"! pip install -q tables  # needed for loading HDF files","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-20T22:54:40.554399Z","iopub.execute_input":"2022-08-20T22:54:40.554918Z","iopub.status.idle":"2022-08-20T22:54:55.950774Z","shell.execute_reply.started":"2022-08-20T22:54:40.554783Z","shell.execute_reply":"2022-08-20T22:54:55.949130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\n\nimport os\nimport torch\nimport numpy as np\nimport pandas as pd\nimport dask.dataframe as dd\nimport matplotlib.pyplot as plt\n\nPATH_DATASET = \"/kaggle/input/open-problems-multimodal\"","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:54:55.954120Z","iopub.execute_input":"2022-08-20T22:54:55.955036Z","iopub.status.idle":"2022-08-20T22:54:58.499851Z","shell.execute_reply.started":"2022-08-20T22:54:55.954976Z","shell.execute_reply":"2022-08-20T22:54:58.498364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Browsing the matadata","metadata":{}},{"cell_type":"code","source":"df_meta = pd.read_csv(os.path.join(PATH_DATASET, \"metadata.csv\")).set_index(\"cell_id\")\ndisplay(df_meta.head())\n\nprint(f\"table size: {len(df_meta)}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:54:58.501832Z","iopub.execute_input":"2022-08-20T22:54:58.502250Z","iopub.status.idle":"2022-08-20T22:54:58.969495Z","shell.execute_reply.started":"2022-08-20T22:54:58.502213Z","shell.execute_reply":"2022-08-20T22:54:58.968229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axarr = plt.subplots(nrows=1, ncols=3, figsize=(12, 5))\nfor i, col in enumerate([\"day\", \"donor\", \"technology\"]):\n    _= df_meta[[col]].value_counts().plot.pie(ax=axarr[i], autopct='%1.1f%%', ylabel=col)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:54:58.972614Z","iopub.execute_input":"2022-08-20T22:54:58.973792Z","iopub.status.idle":"2022-08-20T22:54:59.390090Z","shell.execute_reply.started":"2022-08-20T22:54:58.973750Z","shell.execute_reply":"2022-08-20T22:54:59.388875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axarr = plt.subplots(nrows=3, ncols=1, figsize=(8, 12))\nfor i, col in enumerate([\"cell_type\", \"day\", \"technology\"]):\n    _= df_meta.groupby([col, 'donor']).size().unstack().plot(\n        ax=axarr[i], kind='bar', stacked=True, grid=True\n    )","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:54:59.392053Z","iopub.execute_input":"2022-08-20T22:54:59.392808Z","iopub.status.idle":"2022-08-20T22:55:00.098768Z","shell.execute_reply.started":"2022-08-20T22:54:59.392764Z","shell.execute_reply":"2022-08-20T22:55:00.097379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axarr = plt.subplots(nrows=3, ncols=1, figsize=(8, 12))\nfor i, col in enumerate([\"cell_type\", \"donor\", \"technology\"]):\n    _= df_meta.groupby([col, 'day']).size().unstack().plot(\n        ax=axarr[i], kind='bar', stacked=True, grid=True\n    )","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:55:00.100171Z","iopub.execute_input":"2022-08-20T22:55:00.100512Z","iopub.status.idle":"2022-08-20T22:55:01.021465Z","shell.execute_reply.started":"2022-08-20T22:55:00.100481Z","shell.execute_reply":"2022-08-20T22:55:01.020228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Browse the train dataset","metadata":{}},{"cell_type":"code","source":"df_cite = pd.read_hdf(os.path.join(PATH_DATASET, \"train_cite_inputs.h5\")).astype(np.float16)\ncols_source = list(df_cite.columns)\ndisplay(df_cite.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:59:36.037218Z","iopub.execute_input":"2022-08-20T22:59:36.037657Z","iopub.status.idle":"2022-08-20T23:00:40.727312Z","shell.execute_reply.started":"2022-08-20T22:59:36.037624Z","shell.execute_reply":"2022-08-20T23:00:40.725598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols1, cols2 = list(zip(*[c.split(\"_\") for c in df_cite.columns]))\ncols1 = sorted(set(cols1))\ncols2 = sorted(set(cols2))\nprint(f\"cols1: {len(cols1)}\")\nprint(f\"cols2: {len(cols2)}\")\n\nmx = np.zeros((len(cols1), len(cols2)))\nspl = df_cite.sample(1)\nfor k, v in dict(spl).items():\n    c1, c2 = k.split(\"_\")\n    mx[cols1.index(c1), cols2.index(c2)] = v\nplt.imshow(mx)\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-08-20T23:14:46.174652Z","iopub.execute_input":"2022-08-20T23:14:46.175098Z","iopub.status.idle":"2022-08-20T23:15:16.011903Z","shell.execute_reply.started":"2022-08-20T23:14:46.175063Z","shell.execute_reply":"2022-08-20T23:15:16.010624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_hdf(os.path.join(PATH_DATASET, \"train_cite_targets.h5\")).astype(np.float16)\ncols_target = list(df.columns)\ndisplay(df.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:56:06.284500Z","iopub.execute_input":"2022-08-20T22:56:06.284852Z","iopub.status.idle":"2022-08-20T22:56:07.097169Z","shell.execute_reply.started":"2022-08-20T22:56:06.284819Z","shell.execute_reply":"2022-08-20T22:56:07.096053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cite = df_cite.join(df, how='right')\ndf_cite = df_cite.join(df_meta, how=\"left\")\ndel df\n\nprint(f\"total: {len(df_cite)}\")\nprint(f\"cell_id: {len(df_cite)}\")\ndisplay(df_cite.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:56:07.098609Z","iopub.execute_input":"2022-08-20T22:56:07.098936Z","iopub.status.idle":"2022-08-20T22:56:22.376450Z","shell.execute_reply.started":"2022-08-20T22:56:07.098907Z","shell.execute_reply":"2022-08-20T22:56:22.375330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Meta-data details","metadata":{}},{"cell_type":"code","source":"fig, axarr = plt.subplots(nrows=1, ncols=3, figsize=(12, 4))\nfor i, col in enumerate([\"day\", \"donor\", \"cell_type\"]):\n    _= df_cite[[col]].value_counts().plot.pie(ax=axarr[i], autopct='%1.1f%%', ylabel=col)","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:56:22.379910Z","iopub.execute_input":"2022-08-20T22:56:22.380430Z","iopub.status.idle":"2022-08-20T22:56:24.832680Z","shell.execute_reply.started":"2022-08-20T22:56:22.380394Z","shell.execute_reply":"2022-08-20T22:56:24.831342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axarr = plt.subplots(nrows=2, ncols=1, figsize=(8, 8))\nfor i, col in enumerate([\"day\", \"cell_type\"]):\n    _= df_cite.groupby([col, 'donor']).size().unstack().plot(\n        ax=axarr[i], kind='bar', stacked=True, grid=True\n    )","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:56:24.835279Z","iopub.execute_input":"2022-08-20T22:56:24.836729Z","iopub.status.idle":"2022-08-20T22:56:25.258457Z","shell.execute_reply.started":"2022-08-20T22:56:24.836689Z","shell.execute_reply":"2022-08-20T22:56:25.257298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_cite","metadata":{"execution":{"iopub.status.busy":"2022-08-20T22:56:25.260008Z","iopub.execute_input":"2022-08-20T22:56:25.260633Z","iopub.status.idle":"2022-08-20T22:56:25.275521Z","shell.execute_reply.started":"2022-08-20T22:56:25.260597Z","shell.execute_reply":"2022-08-20T22:56:25.273934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Just a fraction of Multi dataset\n\nNote that this dataset is too large to be leaded directly in DataFrame and crashes on Kaggle kernel","metadata":{}},{"cell_type":"code","source":"path_h5 = os.path.join(PATH_DATASET, \"train_multi_inputs.h5\")\ndf_multi_ = pd.read_hdf(path_h5, start=0, stop=100)\ndisplay(df_multi_.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-20T23:16:00.836721Z","iopub.execute_input":"2022-08-20T23:16:00.837976Z","iopub.status.idle":"2022-08-20T23:16:02.174289Z","shell.execute_reply.started":"2022-08-20T23:16:00.837932Z","shell.execute_reply":"2022-08-20T23:16:02.173093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols1, cols2 = list(zip(*[c.split(\":\") for c in df_multi_.columns]))\ncols1 = sorted(set(cols1))\ncols2 = sorted(set(cols2))\nprint(f\"cols1: {len(cols1)}\")\nprint(f\"cols2: {len(cols2)}\")\ndel df_multi_","metadata":{"execution":{"iopub.status.busy":"2022-08-20T23:20:42.091008Z","iopub.execute_input":"2022-08-20T23:20:42.092174Z","iopub.status.idle":"2022-08-20T23:20:43.470123Z","shell.execute_reply.started":"2022-08-20T23:20:42.092098Z","shell.execute_reply":"2022-08-20T23:20:43.468934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_h5 = os.path.join(PATH_DATASET, \"train_multi_targets.h5\")\ndisplay(pd.read_hdf(path_h5, start=0, stop=100).head())","metadata":{"execution":{"iopub.status.busy":"2022-08-18T11:56:01.919051Z","iopub.execute_input":"2022-08-18T11:56:01.919797Z","iopub.status.idle":"2022-08-18T11:56:02.283269Z","shell.execute_reply.started":"2022-08-18T11:56:01.919763Z","shell.execute_reply":"2022-08-18T11:56:02.282130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load all indexes from chunks","metadata":{}},{"cell_type":"code","source":"cell_id = []\nfor i in range(20):\n    path_h5 = os.path.join(PATH_DATASET, \"train_multi_targets.h5\")\n    df = pd.read_hdf(path_h5, start=i * 10000, stop=(i+1) * 10000)\n    print(i, len(df), df[\"ENSG00000121410\"].mean())\n    if len(df) == 0:\n        break\n    cell_id += list(df.index)\n\ndf_multi_ = pd.DataFrame({\"cell_id\": cell_id}).set_index(\"cell_id\")\ndisplay(df_multi_.head())","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-18T11:56:02.284767Z","iopub.execute_input":"2022-08-18T11:56:02.285692Z","iopub.status.idle":"2022-08-18T11:57:08.327482Z","shell.execute_reply.started":"2022-08-18T11:56:02.285627Z","shell.execute_reply":"2022-08-18T11:57:08.326422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_multi_ = df_multi_.join(df_meta, how=\"left\")\n\nprint(f\"total: {len(df_multi_)}\")\nprint(f\"cell_id: {len(df_multi_)}\")\ndisplay(df_multi_.head())","metadata":{"execution":{"iopub.status.busy":"2022-08-18T11:57:08.328814Z","iopub.execute_input":"2022-08-18T11:57:08.329126Z","iopub.status.idle":"2022-08-18T11:57:08.406434Z","shell.execute_reply.started":"2022-08-18T11:57:08.329097Z","shell.execute_reply":"2022-08-18T11:57:08.405298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Meta-data details","metadata":{}},{"cell_type":"code","source":"fig, axarr = plt.subplots(nrows=1, ncols=3, figsize=(12, 4))\nfor i, col in enumerate([\"day\", \"donor\", \"cell_type\"]):\n    _= df_multi_[[col]].value_counts().plot.pie(ax=axarr[i], autopct='%1.1f%%', ylabel=col)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T11:57:08.407718Z","iopub.execute_input":"2022-08-18T11:57:08.408490Z","iopub.status.idle":"2022-08-18T11:57:08.761970Z","shell.execute_reply.started":"2022-08-18T11:57:08.408451Z","shell.execute_reply":"2022-08-18T11:57:08.760153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axarr = plt.subplots(nrows=2, ncols=1, figsize=(8, 8))\nfor i, col in enumerate([\"day\", \"cell_type\"]):\n    _= df_multi_.groupby([col, 'donor']).size().unstack().plot(\n        ax=axarr[i], kind='bar', stacked=True, grid=True\n    )","metadata":{"execution":{"iopub.status.busy":"2022-08-18T11:57:08.763646Z","iopub.execute_input":"2022-08-18T11:57:08.764836Z","iopub.status.idle":"2022-08-18T11:57:09.194155Z","shell.execute_reply.started":"2022-08-18T11:57:08.764793Z","shell.execute_reply":"2022-08-18T11:57:09.193097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Show Evaluation table","metadata":{}},{"cell_type":"code","source":"df_eval = pd.read_csv(os.path.join(PATH_DATASET, \"evaluation_ids.csv\")).set_index(\"row_id\")\ndisplay(df_eval.head())\n\nprint(f\"total: {len(df_eval)}\")\nprint(f\"cell_id: {len(df_eval['cell_id'].unique())}\")\nprint(f\"gene_id: {len(df_eval['gene_id'].unique())}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-18T11:57:09.195481Z","iopub.execute_input":"2022-08-18T11:57:09.195849Z","iopub.status.idle":"2022-08-18T11:58:20.772188Z","shell.execute_reply.started":"2022-08-18T11:57:09.195818Z","shell.execute_reply":"2022-08-18T11:58:20.770804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**NOTE** that this evaluation expect you to run predictions on the both datasets: cite & multi (as you can see bellow)\n\ntarget columns for:\n- **cite**: 140 columns\n- **multi**: 23418 columns","metadata":{}},{"cell_type":"code","source":"! head ../input/open-problems-multimodal/sample_submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-08-18T11:58:20.773639Z","iopub.execute_input":"2022-08-18T11:58:20.774319Z","iopub.status.idle":"2022-08-18T11:58:21.966396Z","shell.execute_reply.started":"2022-08-18T11:58:20.774283Z","shell.execute_reply":"2022-08-18T11:58:21.965293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Statistic predictions 🏴‍ gene means","metadata":{}},{"cell_type":"code","source":"path_h5 = os.path.join(PATH_DATASET, \"train_cite_targets.h5\")\ncol_means = dict(pd.read_hdf(path_h5).mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-18T11:58:21.967914Z","iopub.execute_input":"2022-08-18T11:58:21.968328Z","iopub.status.idle":"2022-08-18T11:58:22.822667Z","shell.execute_reply.started":"2022-08-18T11:58:21.968294Z","shell.execute_reply":"2022-08-18T11:58:22.821416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_sums = []\ncount = 0\nfor i in range(20):\n    path_h5 = os.path.join(PATH_DATASET, \"train_multi_targets.h5\")\n    df = pd.read_hdf(path_h5, start=i * 10000, stop=(i+1) * 10000)\n    count += len(df)\n    if len(df) == 0:\n        break\n    col_sums.append(dict(df.sum()))\n\ndf_multi_ = pd.DataFrame(col_sums)\ndisplay(df_multi_)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-18T11:58:22.824098Z","iopub.execute_input":"2022-08-18T11:58:22.824409Z","iopub.status.idle":"2022-08-18T11:59:37.566374Z","shell.execute_reply.started":"2022-08-18T11:58:22.824363Z","shell.execute_reply":"2022-08-18T11:59:37.565090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_means.update(dict(df_multi_.sum() / count))","metadata":{"execution":{"iopub.status.busy":"2022-08-18T12:00:04.967951Z","iopub.execute_input":"2022-08-18T12:00:04.968345Z","iopub.status.idle":"2022-08-18T12:00:05.140115Z","shell.execute_reply.started":"2022-08-18T12:00:04.968315Z","shell.execute_reply":"2022-08-18T12:00:05.138957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Map target to eval. table","metadata":{}},{"cell_type":"code","source":"df_eval[\"target\"] = df_eval[\"gene_id\"].map(col_means)\ndisplay(df_eval)","metadata":{"execution":{"iopub.status.busy":"2022-08-18T12:00:08.697131Z","iopub.execute_input":"2022-08-18T12:00:08.698208Z","iopub.status.idle":"2022-08-18T12:00:19.653584Z","shell.execute_reply.started":"2022-08-18T12:00:08.698156Z","shell.execute_reply":"2022-08-18T12:00:19.652399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Finalize submission","metadata":{}},{"cell_type":"code","source":"df_eval[[\"target\"]].round(6).to_csv(\"submission.csv\")\n\n! ls -lh .\n! head submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-08-18T12:04:53.072798Z","iopub.execute_input":"2022-08-18T12:04:53.073165Z","iopub.status.idle":"2022-08-18T12:06:57.653564Z","shell.execute_reply.started":"2022-08-18T12:04:53.073137Z","shell.execute_reply":"2022-08-18T12:06:57.652112Z"},"trusted":true},"execution_count":null,"outputs":[]}]}