{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30587,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-19T07:43:38.807814Z","iopub.execute_input":"2023-11-19T07:43:38.808230Z","iopub.status.idle":"2023-11-19T07:43:39.344734Z","shell.execute_reply.started":"2023-11-19T07:43:38.808179Z","shell.execute_reply":"2023-11-19T07:43:39.342749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing libraries ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import Ridge, LinearRegression\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import RepeatedKFold\nfrom sklearn import metrics","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:44:10.402824Z","iopub.execute_input":"2023-11-19T07:44:10.404587Z","iopub.status.idle":"2023-11-19T07:44:11.558298Z","shell.execute_reply.started":"2023-11-19T07:44:10.404518Z","shell.execute_reply":"2023-11-19T07:44:11.557237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_data=pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\n#adata_train=pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/adata_train.parquet', engine='pyarrow')\ndetrain_data=pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet', engine='pyarrow')\n\n#adata_train","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:44:31.045589Z","iopub.execute_input":"2023-11-19T07:44:31.046153Z","iopub.status.idle":"2023-11-19T07:44:33.845491Z","shell.execute_reply.started":"2023-11-19T07:44:31.046114Z","shell.execute_reply":"2023-11-19T07:44:33.843850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detrain_data.insert(0, column=\"id\", value=id_data[\"id\"])\ndetrain_data","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:44:41.689772Z","iopub.execute_input":"2023-11-19T07:44:41.690203Z","iopub.status.idle":"2023-11-19T07:44:41.752044Z","shell.execute_reply.started":"2023-11-19T07:44:41.690167Z","shell.execute_reply":"2023-11-19T07:44:41.750831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detrain_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:44:56.853734Z","iopub.execute_input":"2023-11-19T07:44:56.854171Z","iopub.status.idle":"2023-11-19T07:44:56.863230Z","shell.execute_reply.started":"2023-11-19T07:44:56.854140Z","shell.execute_reply":"2023-11-19T07:44:56.861798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detrain_data.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:45:06.133297Z","iopub.execute_input":"2023-11-19T07:45:06.134015Z","iopub.status.idle":"2023-11-19T07:45:06.191596Z","shell.execute_reply.started":"2023-11-19T07:45:06.133958Z","shell.execute_reply":"2023-11-19T07:45:06.190302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detrain_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:45:14.471387Z","iopub.execute_input":"2023-11-19T07:45:14.471919Z","iopub.status.idle":"2023-11-19T07:45:14.527656Z","shell.execute_reply.started":"2023-11-19T07:45:14.471885Z","shell.execute_reply":"2023-11-19T07:45:14.526256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detrain_data.id.fillna(method=\"ffill\", inplace=True)\ndetrain_data.isnull().sum().sum()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:45:22.244230Z","iopub.execute_input":"2023-11-19T07:45:22.244682Z","iopub.status.idle":"2023-11-19T07:45:22.300047Z","shell.execute_reply.started":"2023-11-19T07:45:22.244632Z","shell.execute_reply":"2023-11-19T07:45:22.298778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detrain_data.describe()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:45:30.900442Z","iopub.execute_input":"2023-11-19T07:45:30.901924Z","iopub.status.idle":"2023-11-19T07:46:16.305498Z","shell.execute_reply.started":"2023-11-19T07:45:30.901872Z","shell.execute_reply":"2023-11-19T07:46:16.304132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lmplot(x=\"id\", y=\"A1BG\", data=detrain_data)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:46:27.872735Z","iopub.execute_input":"2023-11-19T07:46:27.873207Z","iopub.status.idle":"2023-11-19T07:46:28.470169Z","shell.execute_reply.started":"2023-11-19T07:46:27.873171Z","shell.execute_reply":"2023-11-19T07:46:28.468743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features=detrain_data.iloc[:,0]\nlabels=detrain_data.iloc[:, 6:]\nlabels","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:46:36.731902Z","iopub.execute_input":"2023-11-19T07:46:36.732436Z","iopub.status.idle":"2023-11-19T07:46:36.834800Z","shell.execute_reply.started":"2023-11-19T07:46:36.732395Z","shell.execute_reply":"2023-11-19T07:46:36.832777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_features, test_features, train_labels, test_labels = train_test_split(features, labels, test_size = 0.415, random_state = 42)\nprint('Training Features Shape:', train_features.shape)\nprint('Training Labels Shape:', train_labels.shape)\nprint('Testing Features Shape:', test_features.shape)\nprint('Testing Labels Shape:', test_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:46:47.015507Z","iopub.execute_input":"2023-11-19T07:46:47.016095Z","iopub.status.idle":"2023-11-19T07:46:47.125702Z","shell.execute_reply.started":"2023-11-19T07:46:47.016052Z","shell.execute_reply":"2023-11-19T07:46:47.124105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train the data","metadata":{}},{"cell_type":"code","source":"model = Ridge(alpha=1.0)\n# define model evaluation method\nmodel.fit(train_features.to_numpy().reshape(-1,1), train_labels)\ncv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\n# evaluate model\nscores = cross_val_score(model, train_features.to_numpy().reshape(-1,1), train_labels, scoring='neg_mean_absolute_error', cv=cv, n_jobs=-1)\n# force scores to be positive\nscores = np.absolute(scores)\nprint('Mean MAE: %.3f (%.3f)' % (np.mean(scores), np.std(scores)))\nyhat = model.predict(test_features.to_numpy().reshape(-1,1))\nmetrics.r2_score(test_labels, yhat)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:47:00.619023Z","iopub.execute_input":"2023-11-19T07:47:00.619496Z","iopub.status.idle":"2023-11-19T07:47:10.901075Z","shell.execute_reply.started":"2023-11-19T07:47:00.619463Z","shell.execute_reply":"2023-11-19T07:47:10.899565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building a relationship between gene names and r2_score","metadata":{}},{"cell_type":"code","source":"colnames=test_labels.columns\npredicted_data=pd.DataFrame(data=yhat, columns=colnames)\n\nr2_score=[]\nfor i in range (0,18211):\n    r2_score.append(metrics.r2_score(test_labels.iloc[:,i], predicted_data.iloc[:,i]))","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:47:44.390003Z","iopub.execute_input":"2023-11-19T07:47:44.390457Z","iopub.status.idle":"2023-11-19T07:47:57.961166Z","shell.execute_reply.started":"2023-11-19T07:47:44.390424Z","shell.execute_reply":"2023-11-19T07:47:57.959871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"r2_score_data=pd.DataFrame(data=colnames, columns=[\"Gene_names\"])\nr2_score_data[\"r2_score\"]=r2_score\nmore_accurate_to_predict_genes=r2_score_data[r2_score_data.r2_score>0]","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:48:12.861133Z","iopub.execute_input":"2023-11-19T07:48:12.861621Z","iopub.status.idle":"2023-11-19T07:48:12.879571Z","shell.execute_reply.started":"2023-11-19T07:48:12.861584Z","shell.execute_reply":"2023-11-19T07:48:12.878128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The 1591 genes have positive r2 score than the other genes, thus they have a greater accurate prediction with Ridge regression model than the genes which has negative r2 score.","metadata":{}},{"cell_type":"code","source":"genes_with_greater_r2_score=r2_score_data[r2_score_data.r2_score>0.02]\ngenes_with_greater_r2_score","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:48:31.922401Z","iopub.execute_input":"2023-11-19T07:48:31.922978Z","iopub.status.idle":"2023-11-19T07:48:31.940868Z","shell.execute_reply.started":"2023-11-19T07:48:31.922933Z","shell.execute_reply":"2023-11-19T07:48:31.939569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(9,7))\nsns.barplot(x=\"Gene_names\", y=\"r2_score\", data=genes_with_greater_r2_score)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:48:40.213511Z","iopub.execute_input":"2023-11-19T07:48:40.214005Z","iopub.status.idle":"2023-11-19T07:48:40.472453Z","shell.execute_reply.started":"2023-11-19T07:48:40.213971Z","shell.execute_reply":"2023-11-19T07:48:40.471279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train size of 70% and test size of 30%","metadata":{}},{"cell_type":"code","source":"train_features, test_features, train_labels, test_labels = train_test_split(features, labels, test_size = 0.30, random_state = 42)\nprint('Training Features Shape:', train_features.shape)\nprint('Training Labels Shape:', train_labels.shape)\nprint('Testing Features Shape:', test_features.shape)\nprint('Testing Labels Shape:', test_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:49:12.602298Z","iopub.execute_input":"2023-11-19T07:49:12.602814Z","iopub.status.idle":"2023-11-19T07:49:12.681572Z","shell.execute_reply.started":"2023-11-19T07:49:12.602767Z","shell.execute_reply":"2023-11-19T07:49:12.680051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Ridge(alpha=1.0)\n# define model evaluation method\nmodel.fit(train_features.to_numpy().reshape(-1,1), train_labels)\ncv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\n# evaluate model\nscores = cross_val_score(model, train_features.to_numpy().reshape(-1,1), train_labels, scoring='neg_mean_absolute_error', cv=cv, n_jobs=-1)\n# force scores to be positive\nscores = np.absolute(scores)\nprint('Mean MAE: %.3f (%.3f)' % (np.mean(scores), np.std(scores)))\nyhat_1 = model.predict(test_features.to_numpy().reshape(-1,1))\nmetrics.r2_score(test_labels, yhat_1)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:49:31.104253Z","iopub.execute_input":"2023-11-19T07:49:31.104701Z","iopub.status.idle":"2023-11-19T07:49:38.638845Z","shell.execute_reply.started":"2023-11-19T07:49:31.104669Z","shell.execute_reply":"2023-11-19T07:49:38.637484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train size of 85% and test size of 15%","metadata":{}},{"cell_type":"code","source":"train_features, test_features, train_labels, test_labels = train_test_split(features, labels, test_size = 0.15, random_state = 42)\nprint('Training Features Shape:', train_features.shape)\nprint('Training Labels Shape:', train_labels.shape)\nprint('Testing Features Shape:', test_features.shape)\nprint('Testing Labels Shape:', test_labels.shape)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:50:01.280204Z","iopub.execute_input":"2023-11-19T07:50:01.280770Z","iopub.status.idle":"2023-11-19T07:50:01.374141Z","shell.execute_reply.started":"2023-11-19T07:50:01.280729Z","shell.execute_reply":"2023-11-19T07:50:01.372752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Ridge(alpha=1.0)\n# define model evaluation method\nmodel.fit(train_features.to_numpy().reshape(-1,1), train_labels)\ncv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\n# evaluate model\nscores = cross_val_score(model, train_features.to_numpy().reshape(-1,1), train_labels, scoring='neg_mean_absolute_error', cv=cv, n_jobs=-1)\n# force scores to be positive\nscores = np.absolute(scores)\nprint('Mean MAE: %.3f (%.3f)' % (np.mean(scores), np.std(scores)))\nyhat_2 = model.predict(test_features.to_numpy().reshape(-1,1))\nmetrics.r2_score(test_labels, yhat_2)","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:50:19.096186Z","iopub.execute_input":"2023-11-19T07:50:19.096676Z","iopub.status.idle":"2023-11-19T07:50:26.459863Z","shell.execute_reply.started":"2023-11-19T07:50:19.096625Z","shell.execute_reply":"2023-11-19T07:50:26.458515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split_size=[0.15, 0.30, 0.25]\nr2_score=[-0.04938113919781192, -0.031065985835378215, -0.03629336332854935 ]\n\nsns.lineplot(x=split_size, y=r2_score)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:50:32.362683Z","iopub.execute_input":"2023-11-19T07:50:32.363204Z","iopub.status.idle":"2023-11-19T07:50:32.688021Z","shell.execute_reply.started":"2023-11-19T07:50:32.363162Z","shell.execute_reply":"2023-11-19T07:50:32.686259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submitting the predictions","metadata":{}},{"cell_type":"code","source":"colnames=test_labels.columns\noutput=pd.DataFrame(data=yhat, columns=colnames)\noutput[\"id\"]=id_data[\"id\"]\noutput.to_csv(\"submission.csv\", index=False)\noutput.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-19T07:50:58.643921Z","iopub.execute_input":"2023-11-19T07:50:58.644416Z","iopub.status.idle":"2023-11-19T07:51:11.854086Z","shell.execute_reply.started":"2023-11-19T07:50:58.644384Z","shell.execute_reply":"2023-11-19T07:51:11.853176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}