{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# TOC \n### (This notebook is under construction)\n\n---\n\n#### Version 1\n##### - Metadata (Library, Plate, and Compounds) load, check, and counts\n##### - **Data split (LB split) EDA**\n##### - **Compound plate EDA**\n\n#### Version 2\n##### - DE train EDA (cell type, compounds, clustering)\n##### - MRRMSE\n\n#### Version 3\n##### - SMILES draw\n\n#### Version 4\n##### - Naive prediction and submission\n\n#### Version 5, 6\n##### - Test the data split using the LB score","metadata":{}},{"cell_type":"code","source":"!pip install fastcluster\n!pip install rdkit","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:24:15.361246Z","iopub.execute_input":"2023-10-09T06:24:15.361957Z","iopub.status.idle":"2023-10-09T06:24:39.769247Z","shell.execute_reply.started":"2023-10-09T06:24:15.361917Z","shell.execute_reply":"2023-10-09T06:24:39.767897Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nimport gc\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-09T06:24:44.184165Z","iopub.execute_input":"2023-10-09T06:24:44.184546Z","iopub.status.idle":"2023-10-09T06:24:45.410191Z","shell.execute_reply.started":"2023-10-09T06:24:44.184515Z","shell.execute_reply":"2023-10-09T06:24:45.409222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n---\n---","metadata":{}},{"cell_type":"markdown","source":"# Load metadata, de_train, id_map","metadata":{}},{"cell_type":"code","source":"de_train_meta = pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet').iloc[:, :5]\nde_train_meta","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:24:47.331632Z","iopub.execute_input":"2023-10-09T06:24:47.332121Z","iopub.status.idle":"2023-10-09T06:24:49.725260Z","shell.execute_reply.started":"2023-10-09T06:24:47.332089Z","shell.execute_reply":"2023-10-09T06:24:49.724159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/id_map.csv')\nid_map","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:24:49.726925Z","iopub.execute_input":"2023-10-09T06:24:49.727204Z","iopub.status.idle":"2023-10-09T06:24:49.748063Z","shell.execute_reply.started":"2023-10-09T06:24:49.727180Z","shell.execute_reply":"2023-10-09T06:24:49.746973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adata_obs_meta = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/adata_obs_meta.csv')\nadata_obs_meta","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:24:49.749169Z","iopub.execute_input":"2023-10-09T06:24:49.749448Z","iopub.status.idle":"2023-10-09T06:24:50.834763Z","shell.execute_reply.started":"2023-10-09T06:24:49.749424Z","shell.execute_reply":"2023-10-09T06:24:50.833705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n---","metadata":{}},{"cell_type":"markdown","source":"# Library info.\n### 576 tags = 3 donors * 2 plates * 96 wells","metadata":{}},{"cell_type":"code","source":"library_df = adata_obs_meta.drop(['obs_id', 'cell_type'], axis=1).drop_duplicates()\nlibrary_df","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:24:52.619420Z","iopub.execute_input":"2023-10-09T06:24:52.619753Z","iopub.status.idle":"2023-10-09T06:24:52.820534Z","shell.execute_reply.started":"2023-10-09T06:24:52.619728Z","shell.execute_reply":"2023-10-09T06:24:52.819458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### donor vs plate","metadata":{}},{"cell_type":"code","source":"(\n    library_df.groupby(['donor_id', 'plate_name', 'library_id']).size()\n    .reset_index()\n    .drop([0, 'library_id'], axis=1).drop_duplicates()\n    .groupby('donor_id')['plate_name'].apply(lambda x: ', '.join(x))\n    .reset_index()\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:24:54.126238Z","iopub.execute_input":"2023-10-09T06:24:54.127061Z","iopub.status.idle":"2023-10-09T06:24:54.145992Z","shell.execute_reply.started":"2023-10-09T06:24:54.127026Z","shell.execute_reply":"2023-10-09T06:24:54.144850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### plate vs library","metadata":{"execution":{"iopub.status.busy":"2023-09-18T02:48:59.818008Z","iopub.execute_input":"2023-09-18T02:48:59.818402Z","iopub.status.idle":"2023-09-18T02:48:59.824594Z","shell.execute_reply.started":"2023-09-18T02:48:59.818372Z","shell.execute_reply":"2023-09-18T02:48:59.823255Z"}}},{"cell_type":"code","source":"(\n    library_df.groupby(['donor_id', 'plate_name', 'library_id']).size()\n    .reset_index()\n    .drop([0, 'donor_id'], axis=1).drop_duplicates()\n#     .apply(lambda x: x.str.split('_', expand=True)[1].astype(int), axis=1)\n    .apply(lambda row: pd.Series([row['plate_name'], int(row['library_id'].split('_')[1])], index=row.index), axis=1)\n    .groupby('plate_name')['library_id'].apply(lambda x: x.sort_values().tolist())\n#     .groupby('plate_name')['library_id'].apply(lambda x: ', '.join(x.sort_values().astype(str)))\n    .reset_index()\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:24:55.568431Z","iopub.execute_input":"2023-10-09T06:24:55.568935Z","iopub.status.idle":"2023-10-09T06:24:55.596593Z","shell.execute_reply.started":"2023-10-09T06:24:55.568905Z","shell.execute_reply":"2023-10-09T06:24:55.595498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n---","metadata":{}},{"cell_type":"markdown","source":"# Compounds info.\n- https://www.ebi.ac.uk/chebi/init.do\n- http://www.ilincs.org/ilincs/signatures/main\n\n### 147 = 144 + 2 positive + 1 negative","metadata":{}},{"cell_type":"code","source":"compounds_df = library_df.drop(['library_id', 'plate_name', 'well', 'row', 'col', 'cell_id', 'donor_id'], axis=1).drop_duplicates()\ncompounds_df","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:24:57.504441Z","iopub.execute_input":"2023-10-09T06:24:57.505048Z","iopub.status.idle":"2023-10-09T06:24:57.523069Z","shell.execute_reply.started":"2023-10-09T06:24:57.505018Z","shell.execute_reply":"2023-10-09T06:24:57.521841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compounds_df['sm_name'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:24:58.442695Z","iopub.execute_input":"2023-10-09T06:24:58.443691Z","iopub.status.idle":"2023-10-09T06:24:58.451108Z","shell.execute_reply.started":"2023-10-09T06:24:58.443651Z","shell.execute_reply":"2023-10-09T06:24:58.450271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# Data split (LB split)","metadata":{"execution":{"iopub.status.busy":"2023-09-18T03:29:57.650842Z","iopub.execute_input":"2023-09-18T03:29:57.651329Z","iopub.status.idle":"2023-09-18T03:29:57.657324Z","shell.execute_reply.started":"2023-09-18T03:29:57.651296Z","shell.execute_reply":"2023-09-18T03:29:57.656047Z"}}},{"cell_type":"code","source":"control_ids = [\n#     'LSM-36361'  # DMSO\n    'LSM-43181',  # Belinostat\n    'LSM-6303',  # Dabrafenib\n]","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:00.358213Z","iopub.execute_input":"2023-10-09T06:25:00.358822Z","iopub.status.idle":"2023-10-09T06:25:00.362920Z","shell.execute_reply.started":"2023-10-09T06:25:00.358791Z","shell.execute_reply":"2023-10-09T06:25:00.361382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"privte_ids = [\n    'LSM-45710', 'LSM-4062', \n    'LSM-2193',  #  'forskolin' -> 'Colforsin'\n    'LSM-4105', 'LSM-4031', 'LSM-1099', 'LSM-45153', 'LSM-3822', 'LSM-4933', \n    'LSM-45630',  # 'KD-025' -> 'SLx-2119'\n    'LSM-6258', 'LSM-1023', 'LSM-2655', 'LSM-47602', 'LSM-3349', 'LSM-1020', 'LSM-1143',\n    'LSM-3828', 'LSM-1051', 'LSM-1120', 'LSM-5467', 'LSM-2292', 'LSM-43293', 'LSM-45437',\n    'LSM-2703', 'LSM-45831', 'LSM-1179', 'LSM-1199', 'LSM-1190', 'LSM-36374', 'LSM-5215',\n    'LSM-1195', 'LSM-45468', 'LSM-45410', 'LSM-47459', 'LSM-45663', 'LSM-45518', 'LSM-1062',\n    'LSM-3667',  # 'BRD-K74305673' -> 'IMD-0354',\n    'LSM-1032', 'LSM-5855', 'LSM-45988',\n    'LSM-24954',  # 'BRD-K98039984' -> 'Prednisolone'\n    'LSM-6286', 'LSM-45984', 'LSM-1124', 'LSM-1165', 'LSM-42802', 'LSM-1121', 'LSM-6308',\n    'LSM-1136', 'LSM-1186', 'LSM-45915', 'LSM-2621', 'LSM-5341', 'LSM-45724', 'LSM-2219',\n    'LSM-2936', 'LSM-3171', 'LSM-46889', 'LSM-2379', 'LSM-47132', 'LSM-47120', 'LSM-47437',\n    'LSM-1139', 'LSM-1144', 'LSM-4353', 'LSM-1210', 'LSM-5887', 'LSM-1025', 'LSM-5771', 'LSM-1132',\n    'LSM-1263',  # 'BRD-A04553218' -> 'Chlorpheniramine'\n    'LSM-1167',\n    'LSM-1194',  # 'BRD-A92800748' -> 'TIE2 Kinase Inhibitor'\n    'LSM-45948', 'LSM-45514', 'LSM-5430', 'LSM-2309', \n]\n\nlen(privte_ids)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:01.243948Z","iopub.execute_input":"2023-10-09T06:25:01.244316Z","iopub.status.idle":"2023-10-09T06:25:01.253945Z","shell.execute_reply.started":"2023-10-09T06:25:01.244288Z","shell.execute_reply":"2023-10-09T06:25:01.252790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"public_ids = [\n    'LSM-43216', 'LSM-1050', 'LSM-45849', 'LSM-42800', 'LSM-1131', 'LSM-6335', 'LSM-1211',\n    'LSM-45239', 'LSM-1130', 'LSM-45786', 'LSM-5199', 'LSM-45281',\n    'LSM-6324', # 'ACY-1215' -> 'Ricolinostat'\n    'LSM-3309', 'LSM-1056', 'LSM-45591', 'LSM-46203', 'LSM-5662',\n    'LSM-47134',  # 'SB-2342' -> '5-(9-Isopropyl-8-methyl-2-morpholino-9H-purin-6-yl)pyrimidin-2-amine\t'\n    'LSM-45637', 'LSM-1127', 'LSM-46971', 'LSM-1172', 'LSM-46042', 'LSM-1101', 'LSM-45758',\n    'LSM-5218', 'LSM-2287', 'LSM-1014',\n    'LSM-1040', #  'fostamatinib' -> 'Tamatinib'\n    'LSM-1476;LSM-5290',\n    'LSM-45680',  # 'basimglurant' -> 'RG7090'\n    'LSM-4349',  # '5-iodotubercidin' -> 'IN1451'\n    'LSM-3425', 'LSM-45806',\n    'LSM-45616',  # 'SB-683698' -> 'TR-14035'\n    'LSM-1055',\n    'LSM-43281',  # 'C-646' -> 'STK219801'\n    'LSM-5690', 'LSM-1155', 'LSM-2499',\n    'LSM-2382',  # 'JTC-801' -> 'UNII-BXU45ZH6LI'\n    'LSM-45220', 'LSM-1037', 'LSM-1005', 'LSM-1180', 'LSM-36812',\n    'LSM-45924',  # 'filgotinib' -> 'GLPG0634'\n    'LSM-2013',  # 'TL-HRAS-61' -> TL_HRAS26'\n    'LSM-4738',\n]\n\nlen(public_ids)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:01.980094Z","iopub.execute_input":"2023-10-09T06:25:01.980469Z","iopub.status.idle":"2023-10-09T06:25:01.989490Z","shell.execute_reply.started":"2023-10-09T06:25:01.980440Z","shell.execute_reply":"2023-10-09T06:25:01.988380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ids = [\n    'LSM-1027', 'LSM-1071', 'LSM-45916',\n    'LSM-4944',  # 'ixazomib' -> 'MLN 2238'\n    'LSM-47425',  # 'IWP-L6' -> 'Porcn Inhibitor III'\n    'LSM-1115',\n    'LSM-6237',  # 'CD-437' -> 'O-Demethylated Adapalene'\n    'LSM-1205', 'LSM-45574',\n    'LSM-4255',  # 'NVP-BEZ235' -> 'Dactolisib'\n    'LSM-1181', 'LSM-1158', 'LSM-2334', 'LSM-45496', 'LSM-1011',\n]\n\nlen(train_ids)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:02.838051Z","iopub.execute_input":"2023-10-09T06:25:02.838642Z","iopub.status.idle":"2023-10-09T06:25:02.845288Z","shell.execute_reply.started":"2023-10-09T06:25:02.838596Z","shell.execute_reply":"2023-10-09T06:25:02.844465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_type_ids = [\n    'NK cells',\n    'T cells CD4+',\n    'T cells CD8+',\n    'T regulatory cells',\n    'B cells',\n    'Myeloid cells',\n]\n\ncompounds_ids = control_ids + privte_ids + public_ids + train_ids","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:03.593667Z","iopub.execute_input":"2023-10-09T06:25:03.594572Z","iopub.status.idle":"2023-10-09T06:25:03.599115Z","shell.execute_reply.started":"2023-10-09T06:25:03.594540Z","shell.execute_reply":"2023-10-09T06:25:03.598185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"data_split = pd.concat(\n    [\n        pd.merge(de_train_meta, compounds_df, on=['sm_name', 'sm_lincs_id', 'SMILES', 'control'], how='left').assign(data_split='Train'),\n        pd.merge(id_map, compounds_df, on='sm_name', how='left').assign(data_split='test').drop('id', axis=1),\n    ],\n)\n\ndata_split = data_split.set_index(np.arange(data_split.shape[0]))\n\ndata_split.loc[\n    data_split.query('data_split==\"test\"').reset_index().set_index('sm_lincs_id').loc[privte_ids, 'index'], 'data_split'\n] = 'Private Test'\n\ndata_split.loc[\n    data_split.query('data_split==\"test\"').reset_index().set_index('sm_lincs_id').loc[public_ids, 'index'], 'data_split'\n] = 'Public Test'\n\nsplit_fig_df = data_split.pivot(index='cell_type', columns=['sm_lincs_id','sm_name'], values='data_split').loc[cell_type_ids, compounds_ids].fillna('Not observed')\nsplit_fig_df.iloc[:, :2] = 'Control'","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:05.315520Z","iopub.execute_input":"2023-10-09T06:25:05.315920Z","iopub.status.idle":"2023-10-09T06:25:05.362531Z","shell.execute_reply.started":"2023-10-09T06:25:05.315893Z","shell.execute_reply":"2023-10-09T06:25:05.361318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"color_dict = {\n    'Control':'#06306F',\n    'Train': '#0373BB',\n    'Public Test': '#54B0DA',\n    'Private Test': '#C9E0F4',\n    'Not observed' : '#F7FCFF',\n}","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:06.890742Z","iopub.execute_input":"2023-10-09T06:25:06.891126Z","iopub.status.idle":"2023-10-09T06:25:06.895767Z","shell.execute_reply.started":"2023-10-09T06:25:06.891100Z","shell.execute_reply":"2023-10-09T06:25:06.895066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(24, 3))\n\ncbar_ax = fig.add_axes([0.91, 0.3, 0.01, 0.4])\n\nsns.heatmap(\n    split_fig_df.replace(dict(zip(color_dict.keys(), np.arange(len(color_dict))))),\n    cmap=sns.color_palette(color_dict.values()),\n    lw=0.5,\n    cbar_ax=cbar_ax,\n    cbar_kws={\n        'boundaries':np.arange(len(color_dict)+1)-0.5,\n        'ticks':np.arange(len(color_dict)+1)\n    },\n    xticklabels=True, yticklabels=True,\n    ax=ax,\n)\n\ncbar_ax.set_yticklabels(list(color_dict.keys()) + [''])\n\nax.set_xlabel('')\nax.set_ylabel('')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:09.175370Z","iopub.execute_input":"2023-10-09T06:25:09.176430Z","iopub.status.idle":"2023-10-09T06:25:10.782135Z","shell.execute_reply.started":"2023-10-09T06:25:09.176385Z","shell.execute_reply":"2023-10-09T06:25:10.781252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# Compounds plate info.","metadata":{}},{"cell_type":"code","source":"lsm_split_dict = data_split.query('cell_type==\"B cells\" or cell_type == \"Myeloid cells\"')[['sm_lincs_id', 'data_split']].drop_duplicates().set_index('sm_lincs_id')['data_split'].to_dict()\nlsm_split_dict['LSM-36361'] = 'Control'","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:11.980820Z","iopub.execute_input":"2023-10-09T06:25:11.981148Z","iopub.status.idle":"2023-10-09T06:25:11.992672Z","shell.execute_reply.started":"2023-10-09T06:25:11.981123Z","shell.execute_reply":"2023-10-09T06:25:11.991434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plate_A = library_df.query('donor_id == \"donor_0\" and plate_name==\"plate_0\"').pivot(index='row', columns='col', values='sm_lincs_id')\nplate_B = library_df.query('donor_id == \"donor_0\" and plate_name==\"plate_2\"').pivot(index='row', columns='col', values='sm_lincs_id')","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:12.559677Z","iopub.execute_input":"2023-10-09T06:25:12.560038Z","iopub.status.idle":"2023-10-09T06:25:12.578217Z","shell.execute_reply.started":"2023-10-09T06:25:12.560011Z","shell.execute_reply":"2023-10-09T06:25:12.577215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = color_dict.pop('Not observed')","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:12.966484Z","iopub.execute_input":"2023-10-09T06:25:12.967028Z","iopub.status.idle":"2023-10-09T06:25:12.973391Z","shell.execute_reply.started":"2023-10-09T06:25:12.966985Z","shell.execute_reply":"2023-10-09T06:25:12.972479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(ncols=2, figsize=(18, 5))\n\ncbar_ax = fig.add_axes([0.92, 0.3, 0.02, 0.4])\n\nfor ax, plate in zip(axs, [plate_A, plate_B]):\n    sns.heatmap(\n        plate.replace(lsm_split_dict).replace(dict(zip(color_dict.keys(), np.arange(len(color_dict))))),\n        cmap=sns.color_palette(color_dict.values()),\n        fmt='',\n        annot=plate.apply(lambda x: x.str.replace('LSM-', 'LSM-\\n')),\n        lw=2,\n        linecolor='k',\n        xticklabels=True, yticklabels=True,\n        ax=ax,\n        cbar_ax=cbar_ax,\n        cbar_kws={\n            'boundaries':np.arange(len(color_dict)+1)-0.5,\n            'ticks':np.arange(len(color_dict)+1)\n        },        \n    )\n    cbar_ax.set_yticklabels(list(color_dict.keys()) + [''])\n    ax.set_xlabel('')\n    ax.set_ylabel('')\n    ax.tick_params(top=True, bottom=False, labeltop=True, labelbottom=False)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:15.218048Z","iopub.execute_input":"2023-10-09T06:25:15.218393Z","iopub.status.idle":"2023-10-09T06:25:16.564082Z","shell.execute_reply.started":"2023-10-09T06:25:15.218367Z","shell.execute_reply":"2023-10-09T06:25:16.563013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect();","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:16.655592Z","iopub.execute_input":"2023-10-09T06:25:16.655962Z","iopub.status.idle":"2023-10-09T06:25:16.749385Z","shell.execute_reply.started":"2023-10-09T06:25:16.655934Z","shell.execute_reply":"2023-10-09T06:25:16.748083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n---","metadata":{}},{"cell_type":"markdown","source":"# DE train and clustermap","metadata":{}},{"cell_type":"code","source":"de_train = pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet')\nde_train","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:17.561347Z","iopub.execute_input":"2023-10-09T06:25:17.561911Z","iopub.status.idle":"2023-10-09T06:25:18.951286Z","shell.execute_reply.started":"2023-10-09T06:25:17.561881Z","shell.execute_reply":"2023-10-09T06:25:18.950525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_melt_df = de_train.iloc[:, 5:].melt()\nde_melt_df.describe(percentiles=[.01, .05, .1, .25, .5, .75, .9, .95]).drop('count').apply(lambda x:round(x, 3)).T","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:19.012596Z","iopub.execute_input":"2023-10-09T06:25:19.013510Z","iopub.status.idle":"2023-10-09T06:25:21.396764Z","shell.execute_reply.started":"2023-10-09T06:25:19.013480Z","shell.execute_reply":"2023-10-09T06:25:21.395688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_color_dict = dict(zip(de_train['cell_type'].unique(), sns.color_palette('tab10', 6)))\ncompounds_color_dict = dict(zip(de_train['sm_lincs_id'].unique(), sns.color_palette('rainbow', 147)))\ncontrol_color_dict = dict(zip(de_train['control'].unique(), sns.xkcd_palette(['black', 'white'])))\n\ng = sns.clustermap(\n    data=de_train.iloc[:, 5:].clip(lower=-5, upper=5),\n    # method='average',\n    # method='median',\n    method='complete',\n    metric='euclidean',\n    # metric='correlation',\n    # metric='cosine',\n    row_colors=pd.concat(\n        [\n            de_train['cell_type'].apply(lambda x: cell_color_dict[x]),\n            de_train['sm_lincs_id'].apply(lambda x: compounds_color_dict[x]),\n            de_train['control'].apply(lambda x: control_color_dict[x]),\n        ], axis=1\n    ),\n    cmap='RdBu_r',\n    vmin=-2, vmax=2,\n    cbar_pos=(0.03, 0.85, 0.03, 0.12)\n)\n\ng.ax_heatmap.set_xticks([])\ng.ax_heatmap.set_yticks([])\ng.ax_heatmap.set_xlabel('Gene', fontsize=16)\ng.ax_heatmap.set_ylabel('cell_type x compounds', fontsize=16)\ng.ax_row_colors.tick_params(top=True, labeltop=True, bottom=False, labelbottom=False, rotation=90)\ng.ax_col_dendrogram.set_title('de_train: sign(LFC) * -log10(p-value)', fontsize=16)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T05:47:23.478956Z","iopub.execute_input":"2023-10-09T05:47:23.479301Z","iopub.status.idle":"2023-10-09T05:49:14.565663Z","shell.execute_reply.started":"2023-10-09T05:47:23.479274Z","shell.execute_reply":"2023-10-09T05:49:14.564392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n---","metadata":{}},{"cell_type":"markdown","source":"# Evaluation: Mean Rowwise Root Mean Squared Error\n### We are predicting the sign(LFC) * -log10(p-value) \n### The outliers are the goal, they are very important.","metadata":{}},{"cell_type":"code","source":"def mrrmse(y_pred: pd.DataFrame, y_true: pd.DataFrame):\n    return ((y_pred - y_true)**2).mean(axis=1).apply(np.sqrt).mean()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:24.445604Z","iopub.execute_input":"2023-10-09T06:25:24.446492Z","iopub.status.idle":"2023-10-09T06:25:24.452282Z","shell.execute_reply.started":"2023-10-09T06:25:24.446445Z","shell.execute_reply":"2023-10-09T06:25:24.451024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = de_train.iloc[:, 5:]\ny_pred1 = y_true.clip(lower=-5, upper=5)  # force max=5, MRRMSE=0.46\ny_pred2 = y_true.clip(lower=-2, upper=2)  # force max=2, MRRMSE=0.67\ny_pred3 = y_true.apply(lambda x: round(x, 2))  # round,  MRRMSE=0.00\ny_pred4 = y_true*(y_true.abs()>1)  # only keep >1 value, MRRMSE=0.38\ny_pred5 = y_true*(y_true.abs()>2)  # only keep >2 value, MRRMSE=0.60\ny_pred6 = y_true*(y_true.abs()>5)  # only keep >5 value, MRRMSE=0.80","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:35.944885Z","iopub.execute_input":"2023-10-09T06:25:35.945408Z","iopub.status.idle":"2023-10-09T06:25:39.045315Z","shell.execute_reply.started":"2023-10-09T06:25:35.945380Z","shell.execute_reply":"2023-10-09T06:25:39.044327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for y_pred in [y_pred1, y_pred2, y_pred3, y_pred4, y_pred5, y_pred6]:\n    print (mrrmse(y_pred, y_true))","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:45.976862Z","iopub.execute_input":"2023-10-09T06:25:45.977194Z","iopub.status.idle":"2023-10-09T06:25:46.998819Z","shell.execute_reply.started":"2023-10-09T06:25:45.977169Z","shell.execute_reply":"2023-10-09T06:25:46.998046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---\n---\n---","metadata":{}},{"cell_type":"markdown","source":"# SMILES","metadata":{}},{"cell_type":"code","source":"# https://www.rdkit.org/docs/GettingStartedInPython.html\n\nfrom rdkit import Chem\n\nsm = np.random.choice(compounds_df['SMILES'])\nm = Chem.MolFromSmiles(sm)\nChem.Draw.MolToImage(m)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:48.241309Z","iopub.execute_input":"2023-10-09T06:25:48.242343Z","iopub.status.idle":"2023-10-09T06:25:48.357936Z","shell.execute_reply.started":"2023-10-09T06:25:48.242298Z","shell.execute_reply":"2023-10-09T06:25:48.357147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"code","source":"# # https://github.com/MoleculeTransformers/smiles-featurizers\n\n# !pip install farm\n\n# !git clone https://github.com/MoleculeTransformers/smiles-featurizers\n# !sed -i '/torch/d' ./smiles-featurizers/requirements.txt\n# !sed -i '/farm/d' ./smiles-featurizers/requirements.txt\n# !pip install ./smiles-featurizers","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:50.711749Z","iopub.execute_input":"2023-10-09T06:25:50.712094Z","iopub.status.idle":"2023-10-09T06:25:50.716864Z","shell.execute_reply.started":"2023-10-09T06:25:50.712068Z","shell.execute_reply":"2023-10-09T06:25:50.715494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # https://github.com/DeepGraphLearning/torchdrug\n\n# !pip install torchdrug","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:51.477698Z","iopub.execute_input":"2023-10-09T06:25:51.478088Z","iopub.status.idle":"2023-10-09T06:25:51.483063Z","shell.execute_reply.started":"2023-10-09T06:25:51.478058Z","shell.execute_reply":"2023-10-09T06:25:51.481709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"---","metadata":{}},{"cell_type":"markdown","source":"# Naive prediction and submission","metadata":{"execution":{"iopub.status.busy":"2023-09-21T08:03:10.117462Z","iopub.execute_input":"2023-09-21T08:03:10.117900Z","iopub.status.idle":"2023-09-21T08:03:10.127224Z","shell.execute_reply.started":"2023-09-21T08:03:10.117858Z","shell.execute_reply":"2023-09-21T08:03:10.126297Z"}}},{"cell_type":"code","source":"sub_df = pd.merge(\n    pd.merge(\n        pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'),\n        compounds_df,\n        on='sm_name',\n        how='left'\n    ),\n    de_train.drop(['cell_type', 'sm_name', 'SMILES', 'control'], axis=1).groupby('sm_lincs_id').mean(),  # just mean all the cell type\n    left_on='sm_lincs_id',\n    right_index=True,\n    how='left'\n)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:54.227659Z","iopub.execute_input":"2023-10-09T06:25:54.228164Z","iopub.status.idle":"2023-10-09T06:25:54.548221Z","shell.execute_reply.started":"2023-10-09T06:25:54.228128Z","shell.execute_reply":"2023-10-09T06:25:54.547091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.drop(['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'dose_uM', 'timepoint_hr', 'control'], axis=1).to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:25:55.695443Z","iopub.execute_input":"2023-10-09T06:25:55.696514Z","iopub.status.idle":"2023-10-09T06:26:05.082148Z","shell.execute_reply.started":"2023-10-09T06:25:55.696472Z","shell.execute_reply":"2023-10-09T06:26:05.080937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test the data split using LB score","metadata":{}},{"cell_type":"code","source":"sub_split_df = pd.merge(\n    sub_df,\n    data_split,\n    on=['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'control', 'dose_uM', 'timepoint_hr'],\n).set_index('data_split')\n\nsub_split_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:26:05.084180Z","iopub.execute_input":"2023-10-09T06:26:05.085118Z","iopub.status.idle":"2023-10-09T06:26:05.187172Z","shell.execute_reply.started":"2023-10-09T06:26:05.085073Z","shell.execute_reply":"2023-10-09T06:26:05.186164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask_template = sub_split_df.copy().drop(['cell_type', 'sm_name', 'sm_lincs_id', 'SMILES', 'dose_uM', 'timepoint_hr', 'control'], axis=1)\n\nmask_public = pd.DataFrame(columns=mask_template.columns, index=mask_template.index).fillna(1)\nmask_public.loc['Public Test'] = 0\nmask_public['id'] = 1\n\nmask_private = pd.DataFrame(columns=mask_template.columns, index=mask_template.index).fillna(1)\nmask_private.loc['Private Test'] = 0\nmask_private['id'] = 1\n\nmask_all = pd.DataFrame(columns=mask_template.columns, index=mask_template.index).fillna(0)\nmask_all['id'] = 1","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:26:29.577759Z","iopub.execute_input":"2023-10-09T06:26:29.578171Z","iopub.status.idle":"2023-10-09T06:26:38.224097Z","shell.execute_reply.started":"2023-10-09T06:26:29.578143Z","shell.execute_reply":"2023-10-09T06:26:38.222937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(mask_template * mask_public).to_csv('submission_mask_public.csv', index=False)  # LB 0.666\n(mask_template * mask_private).to_csv('submission_mask_private.csv', index=False)  # LB 0.627\n(mask_template * mask_all).to_csv('submission_mask_all.csv', index=False)  # LB 0.666","metadata":{"execution":{"iopub.status.busy":"2023-10-09T06:26:51.794622Z","iopub.execute_input":"2023-10-09T06:26:51.794971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}