{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### This is a very simple baseline approach that assumes drugs (small molecules) cause generally similar transcriptomic responses in different cell types  \n\nWe simply take the average gene expression change per drug and use that value as our prediction! i.e. the predicted values for both B cells and Myeloid cells in this approach for a given gene + drug combination would be the same!\n\nNotebook inspired by that of Alex https://www.kaggle.com/code/alexandervc/op2-eda-baseline-s\n\n## Ensemble\n\nWe also carry out a simple ensemble by taking the average of all predicted expression values with that produced by Alex in the linked notebook above (V18)","metadata":{}},{"cell_type":"code","source":"# Import necessary modules\n\nimport numpy as np \nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:09.573741Z","iopub.execute_input":"2023-10-04T22:05:09.574309Z","iopub.status.idle":"2023-10-04T22:05:09.579801Z","shell.execute_reply.started":"2023-10-04T22:05:09.574264Z","shell.execute_reply":"2023-10-04T22:05:09.578443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_train = pd.read_parquet(train_path)\nprint(df_train.shape)\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:09.581612Z","iopub.execute_input":"2023-10-04T22:05:09.581903Z","iopub.status.idle":"2023-10-04T22:05:11.223191Z","shell.execute_reply.started":"2023-10-04T22:05:09.581878Z","shell.execute_reply":"2023-10-04T22:05:11.222157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#group by drug and take the mean\ndf_simple = df_train.iloc[:, [1] + list(range(5, df_train.shape[1]))]\n\nmean_df = df_simple.groupby('sm_name').mean().reset_index()\n\nmean_df","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:11.224778Z","iopub.execute_input":"2023-10-04T22:05:11.225842Z","iopub.status.idle":"2023-10-04T22:05:11.613798Z","shell.execute_reply.started":"2023-10-04T22:05:11.225801Z","shell.execute_reply":"2023-10-04T22:05:11.613085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:11.61701Z","iopub.execute_input":"2023-10-04T22:05:11.617384Z","iopub.status.idle":"2023-10-04T22:05:11.655113Z","shell.execute_reply.started":"2023-10-04T22:05:11.617357Z","shell.execute_reply":"2023-10-04T22:05:11.654016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#group by drug and take the mean\ndf_simple_org = df_train.iloc[:, [0] + list(range(5, df_train.shape[1]))]\n\nmean_df_org = df_simple_org.groupby('cell_type').mean().reset_index()\n\nmean_df_org","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:11.658351Z","iopub.execute_input":"2023-10-04T22:05:11.659264Z","iopub.status.idle":"2023-10-04T22:05:11.865066Z","shell.execute_reply.started":"2023-10-04T22:05:11.659221Z","shell.execute_reply":"2023-10-04T22:05:11.864015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create submission file","metadata":{}},{"cell_type":"code","source":"# Load in ID map\ntemp_path = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_ids = pd.read_csv(temp_path)\nprint(df_ids.shape)\ndisplay(df_ids)\n\n# Load in the sample template \ntemp_path = '/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'\nsubmit_df = pd.read_csv(temp_path)\nprint(submit_df.shape)\ndisplay(submit_df)\n\nsubmit_df_org = submit_df.copy()","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:11.866467Z","iopub.execute_input":"2023-10-04T22:05:11.867692Z","iopub.status.idle":"2023-10-04T22:05:14.783644Z","shell.execute_reply.started":"2023-10-04T22:05:11.867653Z","shell.execute_reply":"2023-10-04T22:05:14.782616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List to store the filtered rows\nfiltered_rows = []\n\nfor variable_name in df_ids['sm_name']:\n    matching_rows = mean_df[mean_df['sm_name'] == variable_name].copy()\n    matching_rows['variable_name'] = variable_name  # Add the additional column indicating the variable_name\n    filtered_rows.append(matching_rows)\n\n# Concatenate all filtered rows into a single DataFrame\nresult_df = pd.concat(filtered_rows)\nresult_df = result_df.reset_index(drop=True)\nresult_df","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:14.784919Z","iopub.execute_input":"2023-10-04T22:05:14.786198Z","iopub.status.idle":"2023-10-04T22:05:15.649446Z","shell.execute_reply.started":"2023-10-04T22:05:14.786152Z","shell.execute_reply":"2023-10-04T22:05:15.648325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List to store the filtered rows\nfiltered_rows_org = []\n\nfor variable_name in df_ids['cell_type']:\n    matching_rows_org = mean_df_org[mean_df_org['cell_type'] == variable_name].copy()\n    matching_rows_org['variable_name'] = variable_name  # Add the additional column indicating the variable_name\n    filtered_rows_org.append(matching_rows_org)\n\n# Concatenate all filtered rows into a single DataFrame\nresult_df_org = pd.concat(filtered_rows_org)\nresult_df_org = result_df_org.reset_index(drop=True)\nresult_df_org\n ","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:15.651102Z","iopub.execute_input":"2023-10-04T22:05:15.65147Z","iopub.status.idle":"2023-10-04T22:05:16.517486Z","shell.execute_reply.started":"2023-10-04T22:05:15.651436Z","shell.execute_reply":"2023-10-04T22:05:16.51646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We 'paste' in the new results into the sample submission file\nfor i,col in enumerate( submit_df.columns ):\n    if col == 'id':\n        continue\n    submit_df[col] = result_df[col]\n    if (i%1000) == 0: print(i,col)","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:16.518956Z","iopub.execute_input":"2023-10-04T22:05:16.519547Z","iopub.status.idle":"2023-10-04T22:05:26.548499Z","shell.execute_reply.started":"2023-10-04T22:05:16.519511Z","shell.execute_reply":"2023-10-04T22:05:26.547612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Quickly check our submission - make sure it matches the given format!\n\ndisplay(submit_df)","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:26.549561Z","iopub.execute_input":"2023-10-04T22:05:26.550262Z","iopub.status.idle":"2023-10-04T22:05:26.635425Z","shell.execute_reply.started":"2023-10-04T22:05:26.550233Z","shell.execute_reply":"2023-10-04T22:05:26.634335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We 'paste' in the new results into the sample submission file\nfor i,col in enumerate( submit_df_org.columns ):\n    if col == 'id':\n        continue\n    submit_df_org[col] = result_df_org[col]\n    if (i%1000) == 0: print(i,col)","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:05:26.636867Z","iopub.execute_input":"2023-10-04T22:05:26.63785Z","iopub.status.idle":"2023-10-04T22:05:36.423704Z","shell.execute_reply.started":"2023-10-04T22:05:26.637804Z","shell.execute_reply":"2023-10-04T22:05:36.422571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple ensemble","metadata":{}},{"cell_type":"code","source":"\nfn = '/kaggle/input/op2-0616/submission.csv' \nv18sub = pd.read_csv(fn)\nprint(v18sub.shape)\n\n\n\ndf_submit1 = pd.read_csv('/kaggle/input/op2-models-cv-tuning-f2d73e/submission_RidgeTEtsvdLB0612.csv', index_col = 0)\ndf_submit_aggr_compound = pd.read_csv('/kaggle/input/op2-models-cv-tuning-f2d73e/df_submit_aggr_compound.csv', index_col='id')\ndf_submit_aggr_cell_type = pd.read_csv('/kaggle/input/op2-models-cv-tuning-f2d73e/df_submit_aggr_cell_type.csv', index_col='id')\n\n\nsub_chain = pd.read_csv('/kaggle/input/2-op2-regressor-chain/submission.csv', index_col='id')\nsub_import1 = pd.read_csv('/kaggle/input/op2-603/op2_603.csv', index_col='id')\nsub_import2 = pd.read_csv('/kaggle/input/op2-720/op2_720.csv', index_col='id')\n\n#submission= sub_import1 *0.72 + sub_import2 *0.12 + sub_chain *0.16 \\\n#    +0.45*df_submit1 + 0.45 * df_submit_aggr_compound + 0.1 * df_submit_aggr_cell_type \\\n#    +0.45*v18sub +  0.45*submit_df +  0.1*submit_df_org \n#w = [0.72, 0.12, 0.16,\\\n#     0.45, 0.45, 0.1 ,\\\n#     0.45, 0.45, 0.1 ]\nw = [0.720, 0.12, 0.15,\\\n     0.045, 0.045, 0.01 ,\\\n     0.045, 0.045, 0.01 ]\nsubmission= sub_import1 *w[0] + sub_import2 *w[1] + sub_chain *w[2] \\\n    +w[3]*df_submit1 + w[4] * df_submit_aggr_compound + w[5] * df_submit_aggr_cell_type \\\n    +w[6]*v18sub +  w[7]*submit_df +  w[8]*submit_df_org \n\n\n# convert 'id' back to int32 or we get an error \n#submission['id'] = submission['id'].astype('int32')\n\nsubmission = submission.drop(columns='id')\n\n# make csv file\nsubmission.to_csv('submission.csv', index=True)","metadata":{"execution":{"iopub.status.busy":"2023-10-04T22:26:29.715423Z","iopub.execute_input":"2023-10-04T22:26:29.716514Z","iopub.status.idle":"2023-10-04T22:27:29.545876Z","shell.execute_reply.started":"2023-10-04T22:26:29.71643Z","shell.execute_reply":"2023-10-04T22:27:29.544589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Done!\n\nFeel free to leave comments and suggestions on how to improve!","metadata":{}}]}