{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-18T08:58:51.839880Z","iopub.execute_input":"2023-10-18T08:58:51.840243Z","iopub.status.idle":"2023-10-18T08:58:51.849206Z","shell.execute_reply.started":"2023-10-18T08:58:51.840216Z","shell.execute_reply":"2023-10-18T08:58:51.848230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from sklearn.datasets import make_regression\nfrom sklearn import preprocessing","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:58:52.761693Z","iopub.execute_input":"2023-10-18T08:58:52.762703Z","iopub.status.idle":"2023-10-18T08:58:52.766533Z","shell.execute_reply.started":"2023-10-18T08:58:52.762672Z","shell.execute_reply":"2023-10-18T08:58:52.765817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data preparation","metadata":{}},{"cell_type":"code","source":"de_train = pd.read_parquet('../input/open-problems-single-cell-perturbations/de_train.parquet')\noutput_names = de_train.iloc[:,5:].columns.values.tolist()\nde_train","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:58:54.926374Z","iopub.execute_input":"2023-10-18T08:58:54.926804Z","iopub.status.idle":"2023-10-18T08:58:56.842007Z","shell.execute_reply.started":"2023-10-18T08:58:54.926774Z","shell.execute_reply":"2023-10-18T08:58:56.841008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(list(de_train['cell_type'].unique()))\nprint(len(list(de_train['sm_name'].unique())))","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:58:56.844051Z","iopub.execute_input":"2023-10-18T08:58:56.844481Z","iopub.status.idle":"2023-10-18T08:58:56.852098Z","shell.execute_reply.started":"2023-10-18T08:58:56.844442Z","shell.execute_reply":"2023-10-18T08:58:56.850738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"id_map = pd.read_csv('/kaggle/input/open-problems-single-cell-perturbations/id_map.csv',index_col=0)\nprint(list(id_map['cell_type'].unique()))\nprint(len(list(id_map['sm_name'].unique())))","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:58:56.853134Z","iopub.execute_input":"2023-10-18T08:58:56.853413Z","iopub.status.idle":"2023-10-18T08:58:56.872135Z","shell.execute_reply.started":"2023-10-18T08:58:56.853389Z","shell.execute_reply":"2023-10-18T08:58:56.870813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoder = preprocessing.LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:58:56.874293Z","iopub.execute_input":"2023-10-18T08:58:56.874959Z","iopub.status.idle":"2023-10-18T08:58:56.879595Z","shell.execute_reply.started":"2023-10-18T08:58:56.874931Z","shell.execute_reply":"2023-10-18T08:58:56.878912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame()\ndf['cell_type'] = encoder.fit_transform(id_map['cell_type'])\ndf['sm_name'] = encoder.fit_transform(id_map['sm_name'])\n\ndf","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:58:56.880949Z","iopub.execute_input":"2023-10-18T08:58:56.881601Z","iopub.status.idle":"2023-10-18T08:58:56.902889Z","shell.execute_reply.started":"2023-10-18T08:58:56.881574Z","shell.execute_reply":"2023-10-18T08:58:56.902057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_df = pd.DataFrame()\nnum_df['cell_type'] = encoder.fit_transform(de_train['cell_type'])\nnum_df['sm_name'] = encoder.fit_transform(de_train['sm_name'])\nnum_df['sm_lincs_id'] =encoder.fit_transform(de_train['sm_lincs_id'])\nnum_df['smiles'] =encoder.fit_transform(de_train['SMILES'])\nnum_df['control'] = de_train['control'].fillna(False).astype('int')\n\nnum_df","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:58:58.030815Z","iopub.execute_input":"2023-10-18T08:58:58.031162Z","iopub.status.idle":"2023-10-18T08:58:58.054590Z","shell.execute_reply.started":"2023-10-18T08:58:58.031138Z","shell.execute_reply":"2023-10-18T08:58:58.053316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(num_df.iloc[:,:2], de_train.iloc[:,5:], test_size=0.3, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:58:58.592294Z","iopub.execute_input":"2023-10-18T08:58:58.592633Z","iopub.status.idle":"2023-10-18T08:58:58.749163Z","shell.execute_reply.started":"2023-10-18T08:58:58.592607Z","shell.execute_reply":"2023-10-18T08:58:58.747993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create model","metadata":{}},{"cell_type":"code","source":"#from sklearn.ensemble import AdaBoostRegressor\n#from sklearn.multioutput import MultiOutputRegressor\nfrom sklearn.tree import DecisionTreeRegressor\n\nmodel = DecisionTreeRegressor()\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:59:12.438223Z","iopub.execute_input":"2023-10-18T08:59:12.439240Z","iopub.status.idle":"2023-10-18T08:59:13.573618Z","shell.execute_reply.started":"2023-10-18T08:59:12.439202Z","shell.execute_reply":"2023-10-18T08:59:13.572554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train accuracy = {model.score(X_train,y_train) * 100}%\")\nprint(f\"Test accuracy = {model.score(X_test,y_test) * 100}%\")","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:59:28.036957Z","iopub.execute_input":"2023-10-18T08:59:28.037305Z","iopub.status.idle":"2023-10-18T08:59:28.553289Z","shell.execute_reply.started":"2023-10-18T08:59:28.037281Z","shell.execute_reply":"2023-10-18T08:59:28.552159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(df)\npredictions.shape","metadata":{"execution":{"iopub.status.busy":"2023-10-18T08:59:43.632940Z","iopub.execute_input":"2023-10-18T08:59:43.633447Z","iopub.status.idle":"2023-10-18T08:59:43.657736Z","shell.execute_reply.started":"2023-10-18T08:59:43.633407Z","shell.execute_reply":"2023-10-18T08:59:43.656806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Calculate the error ","metadata":{}},{"cell_type":"code","source":"test_out = model.predict(X_test)\nmrrmse_score = np.sqrt(np.square(y_test - test_out).mean(axis=1))","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:00:21.305519Z","iopub.execute_input":"2023-10-18T09:00:21.305896Z","iopub.status.idle":"2023-10-18T09:00:21.366838Z","shell.execute_reply.started":"2023-10-18T09:00:21.305870Z","shell.execute_reply":"2023-10-18T09:00:21.365591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f' RMMSE = {mrrmse_score.mean()}')","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:00:28.147147Z","iopub.execute_input":"2023-10-18T09:00:28.147525Z","iopub.status.idle":"2023-10-18T09:00:28.152897Z","shell.execute_reply.started":"2023-10-18T09:00:28.147495Z","shell.execute_reply":"2023-10-18T09:00:28.151804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.scatter(list(range(0, 185)), mrrmse_score)\nplt.xlabel('Iterations')\nplt.ylabel('MRRMSE')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:00:37.594696Z","iopub.execute_input":"2023-10-18T09:00:37.595053Z","iopub.status.idle":"2023-10-18T09:00:37.832450Z","shell.execute_reply.started":"2023-10-18T09:00:37.595027Z","shell.execute_reply":"2023-10-18T09:00:37.831380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Results ","metadata":{}},{"cell_type":"code","source":"output = pd.DataFrame(predictions, index=id_map.index, columns=output_names)\noutput","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:00:56.206307Z","iopub.execute_input":"2023-10-18T09:00:56.206838Z","iopub.status.idle":"2023-10-18T09:00:56.247973Z","shell.execute_reply.started":"2023-10-18T09:00:56.206797Z","shell.execute_reply":"2023-10-18T09:00:56.246820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-10-18T09:01:04.786130Z","iopub.execute_input":"2023-10-18T09:01:04.786576Z","iopub.status.idle":"2023-10-18T09:01:13.782202Z","shell.execute_reply.started":"2023-10-18T09:01:04.786548Z","shell.execute_reply":"2023-10-18T09:01:13.781195Z"},"trusted":true},"execution_count":null,"outputs":[]}]}