{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30579,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2023-11-13T20:44:11.688141Z","iopub.execute_input":"2023-11-13T20:44:11.688575Z","iopub.status.idle":"2023-11-13T20:44:11.697843Z","shell.execute_reply.started":"2023-11-13T20:44:11.688534Z","shell.execute_reply":"2023-11-13T20:44:11.696689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Objective of the problem","metadata":{}},{"cell_type":"markdown","source":"## <span style=\"color:blue\"> It is important to map causual link between chemical pertubations and downstream effect on cell state. But the experiments involve high costs and are labour intensive. data Science can solve this problem by predicting effect of chemical pertubations on cells and accelerate the development process of new medicines. \n\n![](https://www.shutterstock.com/image-photo/human-cell-molecule-260nw-1188127132.jpg)\n\n## <span style=\"color:blue\"> The objective is to predict effect of chemical pertubations on new cell types that could accelerate the discovery and creation of new medicines to treat or cure diseases. ","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Importing libraries (Software dependencies for building this model)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import Ridge, LinearRegression\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import RepeatedKFold\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:11.699935Z","iopub.execute_input":"2023-11-13T20:44:11.701089Z","iopub.status.idle":"2023-11-13T20:44:11.709257Z","shell.execute_reply.started":"2023-11-13T20:44:11.701020Z","shell.execute_reply":"2023-11-13T20:44:11.707889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## To install the following modules run the following commands in command line in Linux.\n\n## <span style=\"color:green\"> To install pandas:\n\n### <span style=\"color:lightgreen\"> Run pip install pandas \n\n## <span style=\"color:green\"> To install numpy:\n\n### <span style=\"color:lightgreen\"> pip install numpy\n\n## <span style=\"color:green\">To install matplotlib:\n\n### <span style=\"color:lightgreen\">pip install matplotlib\n\n## <span style=\"color:green\">To install seaborn\n\n### <span style=\"color:lightgreen\">pip install seaborn\n\n## <span style=\"color:green\">To install sklearn:\n\n### <span style=\"color:lightgreen\">pip install -U scikit-learn\n\n","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Importing Datasets","metadata":{}},{"cell_type":"code","source":"id_data=pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\n#adata_train=pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/adata_train.parquet', engine='pyarrow')\ndetrain_data=pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet', engine='pyarrow')\n\n#adata_train","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:11.711009Z","iopub.execute_input":"2023-11-13T20:44:11.711422Z","iopub.status.idle":"2023-11-13T20:44:13.194030Z","shell.execute_reply.started":"2023-11-13T20:44:11.711370Z","shell.execute_reply":"2023-11-13T20:44:13.192964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detrain_data.insert(0, column=\"id\", value=id_data[\"id\"])\ndetrain_data","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:13.196473Z","iopub.execute_input":"2023-11-13T20:44:13.196825Z","iopub.status.idle":"2023-11-13T20:44:13.235523Z","shell.execute_reply.started":"2023-11-13T20:44:13.196793Z","shell.execute_reply":"2023-11-13T20:44:13.234409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Explorartory Data Analysis","metadata":{}},{"cell_type":"markdown","source":"## Shape of the dataset","metadata":{}},{"cell_type":"code","source":"detrain_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:13.236783Z","iopub.execute_input":"2023-11-13T20:44:13.237111Z","iopub.status.idle":"2023-11-13T20:44:13.244971Z","shell.execute_reply.started":"2023-11-13T20:44:13.237075Z","shell.execute_reply":"2023-11-13T20:44:13.243706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## There are 614 rows and 18216 columns.","metadata":{}},{"cell_type":"markdown","source":"## Null Values","metadata":{}},{"cell_type":"code","source":"detrain_data.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:13.246700Z","iopub.execute_input":"2023-11-13T20:44:13.247091Z","iopub.status.idle":"2023-11-13T20:44:13.285202Z","shell.execute_reply.started":"2023-11-13T20:44:13.247060Z","shell.execute_reply":"2023-11-13T20:44:13.283696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Accounting for null values","metadata":{}},{"cell_type":"code","source":"detrain_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:13.286690Z","iopub.execute_input":"2023-11-13T20:44:13.287225Z","iopub.status.idle":"2023-11-13T20:44:13.329635Z","shell.execute_reply.started":"2023-11-13T20:44:13.287180Z","shell.execute_reply":"2023-11-13T20:44:13.328392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detrain_data.id.fillna(method=\"ffill\", inplace=True)\ndetrain_data.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:13.331009Z","iopub.execute_input":"2023-11-13T20:44:13.331340Z","iopub.status.idle":"2023-11-13T20:44:13.375239Z","shell.execute_reply.started":"2023-11-13T20:44:13.331310Z","shell.execute_reply":"2023-11-13T20:44:13.373906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The null values has been taken care of.","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Summary of the dataset","metadata":{}},{"cell_type":"code","source":"detrain_data.describe()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:13.377221Z","iopub.execute_input":"2023-11-13T20:44:13.377574Z","iopub.status.idle":"2023-11-13T20:44:48.185457Z","shell.execute_reply.started":"2023-11-13T20:44:13.377543Z","shell.execute_reply":"2023-11-13T20:44:48.184057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Regression Plot","metadata":{}},{"cell_type":"code","source":"sns.lmplot(x=\"id\", y=\"A1BG\", data=detrain_data)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:48.190876Z","iopub.execute_input":"2023-11-13T20:44:48.192159Z","iopub.status.idle":"2023-11-13T20:44:48.721679Z","shell.execute_reply.started":"2023-11-13T20:44:48.192112Z","shell.execute_reply":"2023-11-13T20:44:48.720226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Building the model (Ridge Regressor)","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Select features and labels","metadata":{}},{"cell_type":"code","source":"features=detrain_data.iloc[:,0]\nlabels=detrain_data.iloc[:, 6:]\nlabels","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:48.723435Z","iopub.execute_input":"2023-11-13T20:44:48.723916Z","iopub.status.idle":"2023-11-13T20:44:48.810116Z","shell.execute_reply.started":"2023-11-13T20:44:48.723872Z","shell.execute_reply":"2023-11-13T20:44:48.808819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Split the data","metadata":{}},{"cell_type":"code","source":"train_features, test_features, train_labels, test_labels = train_test_split(features, labels, test_size = 0.415, random_state = 42)\nprint('Training Features Shape:', train_features.shape)\nprint('Training Labels Shape:', train_labels.shape)\nprint('Testing Features Shape:', test_features.shape)\nprint('Testing Labels Shape:', test_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:48.812118Z","iopub.execute_input":"2023-11-13T20:44:48.812585Z","iopub.status.idle":"2023-11-13T20:44:48.901289Z","shell.execute_reply.started":"2023-11-13T20:44:48.812542Z","shell.execute_reply":"2023-11-13T20:44:48.899733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Train the data","metadata":{}},{"cell_type":"code","source":"model = Ridge(alpha=1.0)\n# define model evaluation method\nmodel.fit(train_features.to_numpy().reshape(-1,1), train_labels)\ncv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\n# evaluate model\nscores = cross_val_score(model, train_features.to_numpy().reshape(-1,1), train_labels, scoring='neg_mean_absolute_error', cv=cv, n_jobs=-1)\n# force scores to be positive\nscores = np.absolute(scores)\nprint('Mean MAE: %.3f (%.3f)' % (np.mean(scores), np.std(scores)))\nyhat = model.predict(test_features.to_numpy().reshape(-1,1))\nmetrics.r2_score(test_labels, yhat)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:48.903040Z","iopub.execute_input":"2023-11-13T20:44:48.903537Z","iopub.status.idle":"2023-11-13T20:44:54.135708Z","shell.execute_reply.started":"2023-11-13T20:44:48.903491Z","shell.execute_reply":"2023-11-13T20:44:54.134444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Building a relationship between gene names and r2_score","metadata":{}},{"cell_type":"code","source":"colnames=test_labels.columns\npredicted_data=pd.DataFrame(data=yhat, columns=colnames)\n\nr2_score=[]\nfor i in range (0,18211):\n    r2_score.append(metrics.r2_score(test_labels.iloc[:,i], predicted_data.iloc[:,i]))\n","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:44:54.137031Z","iopub.execute_input":"2023-11-13T20:44:54.137336Z","iopub.status.idle":"2023-11-13T20:45:05.035970Z","shell.execute_reply.started":"2023-11-13T20:44:54.137308Z","shell.execute_reply":"2023-11-13T20:45:05.034942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_score_data=pd.DataFrame(data=colnames, columns=[\"Gene_names\"])\nr2_score_data[\"r2_score\"]=r2_score\nmore_accurate_to_predict_genes=r2_score_data[r2_score_data.r2_score>0]","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:45:05.037413Z","iopub.execute_input":"2023-11-13T20:45:05.037767Z","iopub.status.idle":"2023-11-13T20:45:05.049981Z","shell.execute_reply.started":"2023-11-13T20:45:05.037735Z","shell.execute_reply":"2023-11-13T20:45:05.048576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The 1591 genes have positive r2 score than the other genes, thus they have a greater accurate prediction with Ridge regression model than the genes which has negative r2 score.","metadata":{}},{"cell_type":"code","source":"genes_with_greater_r2_score=r2_score_data[r2_score_data.r2_score>0.02]\ngenes_with_greater_r2_score\n","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:45:05.051839Z","iopub.execute_input":"2023-11-13T20:45:05.052909Z","iopub.status.idle":"2023-11-13T20:45:05.068887Z","shell.execute_reply.started":"2023-11-13T20:45:05.052861Z","shell.execute_reply":"2023-11-13T20:45:05.067631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(9,7))\nsns.barplot(x=\"Gene_names\", y=\"r2_score\", data=genes_with_greater_r2_score)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:45:05.071027Z","iopub.execute_input":"2023-11-13T20:45:05.071849Z","iopub.status.idle":"2023-11-13T20:45:05.276663Z","shell.execute_reply.started":"2023-11-13T20:45:05.071807Z","shell.execute_reply":"2023-11-13T20:45:05.275567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We can see that RABL2B has the highest r2 score (>0.04) followed by the other genes. Thus we can conclude that the effect of cell pertubations on RABL2B gene has been most accurately calculated.","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Changing the size of training and test sets to see how the model performs accordingly","metadata":{}},{"cell_type":"markdown","source":"## Taking a train size of 70% and test size of 30%","metadata":{}},{"cell_type":"code","source":"train_features, test_features, train_labels, test_labels = train_test_split(features, labels, test_size = 0.30, random_state = 42)\nprint('Training Features Shape:', train_features.shape)\nprint('Training Labels Shape:', train_labels.shape)\nprint('Testing Features Shape:', test_features.shape)\nprint('Testing Labels Shape:', test_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:45:05.278379Z","iopub.execute_input":"2023-11-13T20:45:05.278863Z","iopub.status.idle":"2023-11-13T20:45:05.372090Z","shell.execute_reply.started":"2023-11-13T20:45:05.278832Z","shell.execute_reply":"2023-11-13T20:45:05.370455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Ridge(alpha=1.0)\n# define model evaluation method\nmodel.fit(train_features.to_numpy().reshape(-1,1), train_labels)\ncv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\n# evaluate model\nscores = cross_val_score(model, train_features.to_numpy().reshape(-1,1), train_labels, scoring='neg_mean_absolute_error', cv=cv, n_jobs=-1)\n# force scores to be positive\nscores = np.absolute(scores)\nprint('Mean MAE: %.3f (%.3f)' % (np.mean(scores), np.std(scores)))\nyhat_1 = model.predict(test_features.to_numpy().reshape(-1,1))\nmetrics.r2_score(test_labels, yhat_1)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:45:05.374211Z","iopub.execute_input":"2023-11-13T20:45:05.374760Z","iopub.status.idle":"2023-11-13T20:45:10.834068Z","shell.execute_reply.started":"2023-11-13T20:45:05.374709Z","shell.execute_reply":"2023-11-13T20:45:10.832736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Taking a train size of 85% and test size of  15%","metadata":{}},{"cell_type":"code","source":"train_features, test_features, train_labels, test_labels = train_test_split(features, labels, test_size = 0.15, random_state = 42)\nprint('Training Features Shape:', train_features.shape)\nprint('Training Labels Shape:', train_labels.shape)\nprint('Testing Features Shape:', test_features.shape)\nprint('Testing Labels Shape:', test_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:45:10.837083Z","iopub.execute_input":"2023-11-13T20:45:10.837971Z","iopub.status.idle":"2023-11-13T20:45:10.911422Z","shell.execute_reply.started":"2023-11-13T20:45:10.837925Z","shell.execute_reply":"2023-11-13T20:45:10.909949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Ridge(alpha=1.0)\n# define model evaluation method\nmodel.fit(train_features.to_numpy().reshape(-1,1), train_labels)\ncv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\n# evaluate model\nscores = cross_val_score(model, train_features.to_numpy().reshape(-1,1), train_labels, scoring='neg_mean_absolute_error', cv=cv, n_jobs=-1)\n# force scores to be positive\nscores = np.absolute(scores)\nprint('Mean MAE: %.3f (%.3f)' % (np.mean(scores), np.std(scores)))\nyhat_2 = model.predict(test_features.to_numpy().reshape(-1,1))\nmetrics.r2_score(test_labels, yhat_2)","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:45:10.912883Z","iopub.execute_input":"2023-11-13T20:45:10.913251Z","iopub.status.idle":"2023-11-13T20:45:16.496880Z","shell.execute_reply.started":"2023-11-13T20:45:10.913210Z","shell.execute_reply":"2023-11-13T20:45:16.495539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split_size=[0.15, 0.30, 0.25]\nr2_score=[-0.04938113919781192, -0.031065985835378215, -0.03629336332854935 ]\n\nsns.lineplot(x=split_size, y=r2_score)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:45:16.498758Z","iopub.execute_input":"2023-11-13T20:45:16.499114Z","iopub.status.idle":"2023-11-13T20:45:16.716494Z","shell.execute_reply.started":"2023-11-13T20:45:16.499084Z","shell.execute_reply":"2023-11-13T20:45:16.715257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We can see that as split size increses the r2 score value increses on the negative side which equals to a decrease in the r2 score value.","metadata":{}},{"cell_type":"markdown","source":"# <span style=\"color:purple\"> Submitting the predictions","metadata":{}},{"cell_type":"code","source":"colnames=test_labels.columns\noutput=pd.DataFrame(data=yhat, columns=colnames)\noutput[\"id\"]=id_data[\"id\"]\noutput.to_csv(\"submission.csv\", index=False)\noutput.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-13T20:45:16.718697Z","iopub.execute_input":"2023-11-13T20:45:16.719296Z","iopub.status.idle":"2023-11-13T20:45:25.953644Z","shell.execute_reply.started":"2023-11-13T20:45:16.719260Z","shell.execute_reply":"2023-11-13T20:45:25.952152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}