{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":7712331,"sourceType":"datasetVersion","datasetId":4441094}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Solution\n\nThis notebook shows the procedure we used to generate the final submission for the challenge. We first used the median of 146 models (selected by leaving 1 drug out for NK cells) as the base prediction. This model was trained only on the top 64 most variable genes (sorted by variance), and was used to predict the ~18K signed log-pvalues.\n\nWe then trained the same NN on a subset of the data. This produced again 146 models that we used to generate a median prediction only on that subset of the data. \n\nFinally, we replaced the base predictions with the for the final submission.","metadata":{"_uuid":"ada13d1f-eceb-441e-8132-8714df0c1d7a","_cell_guid":"dca61f0f-0371-4dd3-9737-954f608e20fd","trusted":true}},{"cell_type":"code","source":"!pip install gdown\n!gdown 10EI3VgW1_iclzl1J08iGvFbpWjamv9Xj -O data.zip\n!unzip data.zip\n!pip install git+https://github.com/scapeML/scape.git","metadata":{"_uuid":"b88947d6-36e9-4b00-9c55-a2b4c1f77a90","_cell_guid":"9140fc3e-14bd-40da-a768-5b6ed2e886a4","collapsed":false,"execution":{"iopub.status.busy":"2024-03-25T10:01:49.815346Z","iopub.execute_input":"2024-03-25T10:01:49.816606Z","iopub.status.idle":"2024-03-25T10:02:57.036587Z","shell.execute_reply.started":"2024-03-25T10:01:49.816566Z","shell.execute_reply":"2024-03-25T10:02:57.035229Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import scape\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\n\nscape.__version__, pd.__version__, np.__version__, tf.__version__","metadata":{"_uuid":"a52ab189-7a65-493e-bf04-98cb8bf5ef63","_cell_guid":"49ac4c5c-5c34-437a-adbb-0e6d9c2627ea","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-03-24T20:14:50.125603Z","iopub.execute_input":"2024-03-24T20:14:50.126467Z","iopub.status.idle":"2024-03-24T20:15:01.639432Z","shell.execute_reply.started":"2024-03-24T20:14:50.126429Z","shell.execute_reply":"2024-03-24T20:15:01.638502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_de = scape.io.load_slogpvals(\"_data/de_train.parquet\")\ndf_lfc = scape.io.load_lfc(\"_data/lfc_train.parquet\")\n\n# Make sure rows/columns are in the same order\ndf_lfc = df_lfc.loc[df_de.index, df_de.columns]\ndf_de.shape, df_lfc.shape","metadata":{"_uuid":"fcc8b50f-db00-4341-963a-3f78d339f540","_cell_guid":"5a2aa8f6-4d95-46de-a81e-fb50d570ef08","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-03-24T20:15:01.645257Z","iopub.execute_input":"2024-03-24T20:15:01.645856Z","iopub.status.idle":"2024-03-24T20:15:08.159680Z","shell.execute_reply.started":"2024-03-24T20:15:01.645821Z","shell.execute_reply":"2024-03-24T20:15:08.158644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We select only a subset of the genes for the model (top most variant genes)\nn_genes = 64\ntop_genes = scape.util.select_top_variable([df_de], k=n_genes)","metadata":{"_uuid":"ad2d25e6-4046-4fa7-b3b4-e2b948b6d497","_cell_guid":"4db5d03f-99c9-44eb-b040-6009d841bfef","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-03-24T20:15:08.161277Z","iopub.execute_input":"2024-03-24T20:15:08.161740Z","iopub.status.idle":"2024-03-24T20:15:08.308426Z","shell.execute_reply.started":"2024-03-24T20:15:08.161700Z","shell.execute_reply":"2024-03-24T20:15:08.307334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Base predictions","metadata":{"_uuid":"b3c37224-81c9-4d93-8196-9a8abca1f38d","_cell_guid":"a6aba3f4-3565-4ba1-b0d7-4a4dc9502a58","trusted":true}},{"cell_type":"code","source":"cell = \"NK cells\"\ndrugs = df_de.loc[df_de.index.get_level_values(\"cell_type\") == cell].index.get_level_values(\"sm_name\").unique().tolist()\nlen(drugs)","metadata":{"_uuid":"d88be632-3ad7-4dd8-ae8b-0b89d0660661","_cell_guid":"09c20c61-26d9-46e5-b9db-ce7526f8d32f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-03-24T20:15:08.309870Z","iopub.execute_input":"2024-03-24T20:15:08.310221Z","iopub.status.idle":"2024-03-24T20:15:08.330942Z","shell.execute_reply.started":"2024-03-24T20:15:08.310191Z","shell.execute_reply":"2024-03-24T20:15:08.329846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_id_map = pd.read_csv(\"_data/id_map.zip\")\ndf_sub = pd.read_csv(\"_data/sample_submission.zip\", index_col = 0)\ndf_sub_ix = df_id_map.set_index([\"cell_type\", \"sm_name\"])\ndf_sub_ix","metadata":{"_uuid":"7204c727-d9e5-4d7e-a32e-671d50c12534","_cell_guid":"909a0e62-1952-431a-8824-a2242f265637","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-03-24T20:15:08.332234Z","iopub.execute_input":"2024-03-24T20:15:08.332554Z","iopub.status.idle":"2024-03-24T20:15:10.930216Z","shell.execute_reply.started":"2024-03-24T20:15:08.332526Z","shell.execute_reply":"2024-03-24T20:15:10.929324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_predictions = []\nfor i, d in enumerate(drugs):\n    print(i, d)\n    scm = scape.model.create_default_model(n_genes, df_de, df_lfc)\n    result = scm.train(\n        val_cells=[cell], \n        val_drugs=[d],\n        input_columns=top_genes,\n        epochs=300,\n        output_folder=\"_models\",\n        config_file_name=\"config.pkl\",\n        model_file_name=f\"drug{i}.keras\",\n        baselines=[\"zero\", \"slogpval_drug\"],\n    )\n    # Collect prediction in the OOF data\n    df_pred = scm.predict(df_sub_ix)\n    df_pred = df_pred.loc[:, df_sub.columns]\n    base_predictions.append(df_pred)\n\ndf_sub = pd.DataFrame(np.median(base_predictions, axis=0), index=df_sub_ix.index, columns=df_sub.columns)\ndf_sub.to_csv(\"base_predictions.csv\")\ndf_sub","metadata":{"_uuid":"b3dc886a-9579-4a57-886b-d022085903f2","_cell_guid":"93158c0f-20e8-4b53-ae8f-9cf0b330b864","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-03-24T20:15:10.931360Z","iopub.execute_input":"2024-03-24T20:15:10.931665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = df_sub\ndf_submission","metadata":{"_uuid":"c2006e04-625b-489c-b03c-51239619d3b8","_cell_guid":"891f8822-413d-464f-b706-5d18c2162ca7","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission_data = df_sub_ix.join(df_submission).reset_index(drop=True).set_index(\"id\")\ndf_submission_data","metadata":{"_uuid":"55661cd0-9085-400e-98bd-5110db678e77","_cell_guid":"cc9aafd3-0e5f-464f-aafe-d814af9d158d","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission_data.to_csv(\"submission.csv\")","metadata":{"_uuid":"62510572-e5f7-418f-bd91-d0c60d5353ea","_cell_guid":"b76b3b05-6dd0-40a0-8875-c7c7205d7987","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}