{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>\n\n<div class=\"alert alert-success\">  \n    <h1 align=\"center\" style=\"color:darkcyan;\">🧬Open Problems – Single-Cell Perturbations</h1> \n    <h3 align=\"center\" style=\"color:gray;\">Predict how small molecules change gene expression in different cell types</h3> \n    <h3 align=\"center\" style=\"color:gray;\">By: Somayyeh Gholami & Mehran Kazeminia</h3> \n</div>\n\n# <div style=\"color:white;background-color:darkcyan;padding:2%;border-radius:15px 15px;font-size:1em;text-align:center\">Regressor Chain</div>\n\nAdvances in single-cell experimental protocols allow for massively multiplexed experiments to measure hundreds of thousands of cells under thousands of unique conditions. These are commonly termed “perturbations”, which are temporary or permanent changes caused by an external influence [Srivatsan et al., 2020].\n\n![](https://cdn-images-1.medium.com/max/1000/1*jCLUHuVRj4ZiBvLFmd8u5g.png)\n\n**Single-Cell Perturbations** datasets measure how gene expression changes in individual cells when they are exposed to different stimuli such as drugs/chemicals/environmental changes. This information can be used to understand how cells work and how they can be manipulated to treat diseases.\n","metadata":{}},{"cell_type":"code","source":"import warnings # suppress warnings\nwarnings.filterwarnings('ignore')\n#:::::::::::::::::::::::::::::::::::\nimport os\nimport gc\nimport glob\nimport random\nimport numpy as np \nimport pandas as pd\nimport seaborn as sns\nfrom tqdm import tqdm\nfrom scipy import stats\nfrom pathlib import Path\nfrom itertools import groupby\n#:::::::::::::::::::::::::::::::::::\nimport matplotlib.pyplot as plt\nimport plotly.figure_factory as ff\nimport plotly.express as px\n%matplotlib inline\n!ls ../input/*","metadata":{"_kg_hide-input":true,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2023-09-30T10:08:25.488670Z","iopub.execute_input":"2023-09-30T10:08:25.488989Z","iopub.status.idle":"2023-09-30T10:08:28.351686Z","shell.execute_reply.started":"2023-09-30T10:08:25.488966Z","shell.execute_reply":"2023-09-30T10:08:28.349840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>\n\n<div>\n    <h1 align=\"center\" style=\"color:darkred;\">Competition Data (Eight files)</h1>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:darkcyan;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b>(1) de_train.parquet</p></div>","metadata":{}},{"cell_type":"code","source":"de_train = pd.read_parquet('../input/open-problems-single-cell-perturbations/de_train.parquet')\nde_train","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:08:28.353698Z","iopub.execute_input":"2023-09-30T10:08:28.354290Z","iopub.status.idle":"2023-09-30T10:08:30.461910Z","shell.execute_reply.started":"2023-09-30T10:08:28.354265Z","shell.execute_reply":"2023-09-30T10:08:30.460524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:darkcyan;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b>(7) id_map.csv</p></div>","metadata":{}},{"cell_type":"code","source":"id_map = pd.read_csv('../input/open-problems-single-cell-perturbations/id_map.csv')\nid_map","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:08:30.463040Z","iopub.execute_input":"2023-09-30T10:08:30.463262Z","iopub.status.idle":"2023-09-30T10:08:30.491312Z","shell.execute_reply.started":"2023-09-30T10:08:30.463242Z","shell.execute_reply":"2023-09-30T10:08:30.490445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"color:yellow;display:inline-block;border-radius:5px;background-color:darkcyan;font-family:Nexa;overflow:hidden\"><p style=\"padding:15px;color:white;overflow:hidden;font-size:85%;letter-spacing:0.5px;margin:0\"><b> </b>(8) sample_submission.csv</p></div>","metadata":{}},{"cell_type":"code","source":"sample_submission = pd.read_csv('../input/open-problems-single-cell-perturbations/sample_submission.csv', index_col='id')\nsample_submission","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:08:30.493202Z","iopub.execute_input":"2023-09-30T10:08:30.493486Z","iopub.status.idle":"2023-09-30T10:08:32.361979Z","shell.execute_reply.started":"2023-09-30T10:08:30.493442Z","shell.execute_reply":"2023-09-30T10:08:32.360867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>\n\n<div class=\"alert alert-success\">  \n</div>\n\n## <span style=\"color:darkred;\">train | test | target</span>","metadata":{}},{"cell_type":"code","source":"xlist  = ['cell_type','sm_name']\n_ylist = ['cell_type','sm_name','sm_lincs_id','SMILES','control']\n\ny = de_train.drop(columns=_ylist)\ny.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:08:32.363118Z","iopub.execute_input":"2023-09-30T10:08:32.363422Z","iopub.status.idle":"2023-09-30T10:08:32.395637Z","shell.execute_reply.started":"2023-09-30T10:08:32.363395Z","shell.execute_reply":"2023-09-30T10:08:32.394180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## <span style=\"color:darkcyan;\">get_dummies</span>","metadata":{}},{"cell_type":"code","source":"train  = pd.get_dummies(de_train[xlist], columns=xlist)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:08:32.396852Z","iopub.execute_input":"2023-09-30T10:08:32.397110Z","iopub.status.idle":"2023-09-30T10:08:32.433267Z","shell.execute_reply.started":"2023-09-30T10:08:32.397091Z","shell.execute_reply":"2023-09-30T10:08:32.432493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.get_dummies(id_map[xlist], columns=xlist)\ntest.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:08:32.434394Z","iopub.execute_input":"2023-09-30T10:08:32.435503Z","iopub.status.idle":"2023-09-30T10:08:32.449514Z","shell.execute_reply.started":"2023-09-30T10:08:32.435450Z","shell.execute_reply":"2023-09-30T10:08:32.448544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"uncommon = [f for f in train if f not in test]\n\nX = train.drop(columns=uncommon)\nX.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:08:32.450796Z","iopub.execute_input":"2023-09-30T10:08:32.451610Z","iopub.status.idle":"2023-09-30T10:08:32.465889Z","shell.execute_reply.started":"2023-09-30T10:08:32.451583Z","shell.execute_reply":"2023-09-30T10:08:32.464226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>\n\n<div class=\"alert alert-success\">  \n</div>\n\n## <span style=\"color:darkred;\">LinearSVR & MultiOutputRegressor</span>","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import LinearSVR\nfrom sklearn.multioutput import MultiOutputRegressor","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:08:32.467575Z","iopub.execute_input":"2023-09-30T10:08:32.467910Z","iopub.status.idle":"2023-09-30T10:08:32.759330Z","shell.execute_reply.started":"2023-09-30T10:08:32.467881Z","shell.execute_reply":"2023-09-30T10:08:32.758267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = LinearSVR(max_iter= 2000, epsilon= 0.1)\n\nwrapper = MultiOutputRegressor(model)\nwrapper.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:08:32.762093Z","iopub.execute_input":"2023-09-30T10:08:32.762492Z","iopub.status.idle":"2023-09-30T10:09:26.224013Z","shell.execute_reply.started":"2023-09-30T10:08:32.762429Z","shell.execute_reply":"2023-09-30T10:09:26.222784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission1 = pd.DataFrame(wrapper.predict(test), columns= de_train.columns[5:])\nsubmission1.index.name = 'id'\nsubmission1.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:09:26.225280Z","iopub.execute_input":"2023-09-30T10:09:26.225636Z","iopub.status.idle":"2023-09-30T10:09:51.915699Z","shell.execute_reply.started":"2023-09-30T10:09:26.225608Z","shell.execute_reply":"2023-09-30T10:09:51.915046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission1.to_csv('submission1.csv')\n!ls","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:09:51.916721Z","iopub.execute_input":"2023-09-30T10:09:51.917088Z","iopub.status.idle":"2023-09-30T10:09:57.718952Z","shell.execute_reply.started":"2023-09-30T10:09:51.917066Z","shell.execute_reply":"2023-09-30T10:09:57.717544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>\n\n<div class=\"alert alert-success\">  \n</div>\n\n## <span style=\"color:darkred;\">Chained Multioutput Regression</span>\n\nThe first model in the sequence uses the input and predicts one output; the second model uses the input and the output from the first model to make a prediction; the third model uses the input and output from the first two models to make a prediction, and so on.\n\nThis work is usually done for outputs that depend on both the input and the other outputs, and the **RegressorChain** library in sklearn can be used. But in this challenge with about eighteen thousand outputs, it is not possible to use this type of library. But to demonstrate this process, we will only repeat this chain three times and for each time, we will predict only 1000 outputs.","metadata":{}},{"cell_type":"code","source":"y1 = y.iloc[:, :1000].copy()\ny2 = y.iloc[:, 1000:2000].copy()\ny3 = y.iloc[:, 2000:3000].copy()\n\ny.shape, y1.shape, y2.shape, y3.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:09:57.720611Z","iopub.execute_input":"2023-09-30T10:09:57.720907Z","iopub.status.idle":"2023-09-30T10:09:57.732778Z","shell.execute_reply.started":"2023-09-30T10:09:57.720879Z","shell.execute_reply":"2023-09-30T10:09:57.731818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X1 = X.copy()\nX2 = X1.join(y1)\nX3 = X2.join(y2)\n\nX.shape, X1.shape, X2.shape, X3.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:23:29.544225Z","iopub.execute_input":"2023-09-30T10:23:29.544537Z","iopub.status.idle":"2023-09-30T10:23:29.564214Z","shell.execute_reply.started":"2023-09-30T10:23:29.544516Z","shell.execute_reply":"2023-09-30T10:23:29.563027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test1 = test.copy()\ntest1.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:26:58.698650Z","iopub.execute_input":"2023-09-30T10:26:58.698974Z","iopub.status.idle":"2023-09-30T10:26:58.706505Z","shell.execute_reply.started":"2023-09-30T10:26:58.698951Z","shell.execute_reply":"2023-09-30T10:26:58.705771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"code","source":"model = LinearSVR()\nwrapper = MultiOutputRegressor(model)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:40:08.199012Z","iopub.execute_input":"2023-09-30T10:40:08.199368Z","iopub.status.idle":"2023-09-30T10:40:08.204922Z","shell.execute_reply.started":"2023-09-30T10:40:08.199343Z","shell.execute_reply":"2023-09-30T10:40:08.203599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wrapper.fit(X1, y1)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:40:25.661424Z","iopub.execute_input":"2023-09-30T10:40:25.661790Z","iopub.status.idle":"2023-09-30T10:40:28.433767Z","shell.execute_reply.started":"2023-09-30T10:40:25.661763Z","shell.execute_reply":"2023-09-30T10:40:28.432583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pr1 = pd.DataFrame(wrapper.predict(test1), columns= de_train.columns[5:1005])\n\ntest2 = test1.join(pr1)\ntest2.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:40:28.541790Z","iopub.execute_input":"2023-09-30T10:40:28.542105Z","iopub.status.idle":"2023-09-30T10:40:29.895041Z","shell.execute_reply.started":"2023-09-30T10:40:28.542084Z","shell.execute_reply":"2023-09-30T10:40:29.894249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wrapper.fit(X2, y2)","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:41:14.181787Z","iopub.execute_input":"2023-09-30T10:41:14.182088Z","iopub.status.idle":"2023-09-30T10:59:55.834550Z","shell.execute_reply.started":"2023-09-30T10:41:14.182064Z","shell.execute_reply":"2023-09-30T10:59:55.833282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pr2 = pd.DataFrame(wrapper.predict(test2), columns= de_train.columns[1005:2005])\n\ntest3 = test2.join(pr2)\ntest3.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-30T10:59:55.836443Z","iopub.execute_input":"2023-09-30T10:59:55.836751Z","iopub.status.idle":"2023-09-30T11:00:00.398724Z","shell.execute_reply.started":"2023-09-30T10:59:55.836729Z","shell.execute_reply":"2023-09-30T11:00:00.398043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wrapper.fit(X3, y3)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pr3 = pd.DataFrame(wrapper.predict(test3), columns= de_train.columns[2005:3005])\n\ntest4 = test3.join(pr3)\ntest4.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>","metadata":{}},{"cell_type":"code","source":"t = test4.iloc[:, 131:].copy()\nt.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission= t.join(submission1.iloc[:, 3000:])\nsubmission.index.name = 'id'\nsubmission.shape","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv')\n!ls","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-success\">  \n</div>\n\n<div class=\"alert alert-success\">  \n</div>","metadata":{}}]}