{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### This is a very simple baseline approach that assumes drugs (small molecules) cause generally similar transcriptomic responses in different cell types  \n\nWe simply take the average gene expression change per drug and use that value as our prediction! i.e. the predicted values for both B cells and Myeloid cells in this approach for a given gene + drug combination would be the same!\n\nNotebook inspired by that of Alex https://www.kaggle.com/code/alexandervc/op2-eda-baseline-s\n\n## Ensemble\n\nWe also carry out a simple ensemble by taking the average of all predicted expression values with that produced by Alex in the linked notebook above (V18)\n\nFinally we average these predictions with one that takes averaged expression values by cell type. Source : https://www.kaggle.com/code/liudacheldieva/streamlined-baseline-approach/notebook","metadata":{}},{"cell_type":"code","source":"# Import necessary modules\n\nimport numpy as np \nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:50:48.804609Z","iopub.execute_input":"2023-09-17T14:50:48.804982Z","iopub.status.idle":"2023-09-17T14:50:48.810659Z","shell.execute_reply.started":"2023-09-17T14:50:48.804953Z","shell.execute_reply":"2023-09-17T14:50:48.809470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_train = pd.read_parquet(train_path)\nprint(df_train.shape)\ndf_train","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:50:48.812975Z","iopub.execute_input":"2023-09-17T14:50:48.813951Z","iopub.status.idle":"2023-09-17T14:50:50.803476Z","shell.execute_reply.started":"2023-09-17T14:50:48.813921Z","shell.execute_reply":"2023-09-17T14:50:50.802397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#group by drug and take the mean\ndf_simple = df_train.iloc[:, [1] + list(range(5, df_train.shape[1]))]\n\nmean_df = df_simple.groupby('sm_name').mean().reset_index()\n\nmean_df","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:50:50.805393Z","iopub.execute_input":"2023-09-17T14:50:50.805757Z","iopub.status.idle":"2023-09-17T14:50:51.118199Z","shell.execute_reply.started":"2023-09-17T14:50:50.805726Z","shell.execute_reply":"2023-09-17T14:50:51.117173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create submission file","metadata":{}},{"cell_type":"code","source":"# Load in ID map\ntemp_path = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_ids = pd.read_csv(temp_path)\nprint(df_ids.shape)\ndisplay(df_ids)\n\n# Load in the sample template \ntemp_path = '/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'\nsubmit_df = pd.read_csv(temp_path)\nprint(submit_df.shape)\ndisplay(submit_df)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:50:51.119757Z","iopub.execute_input":"2023-09-17T14:50:51.120307Z","iopub.status.idle":"2023-09-17T14:50:55.860681Z","shell.execute_reply.started":"2023-09-17T14:50:51.120278Z","shell.execute_reply":"2023-09-17T14:50:55.859599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List to store the filtered rows\nfiltered_rows = []\n\nfor variable_name in df_ids['sm_name']:\n    matching_rows = mean_df[mean_df['sm_name'] == variable_name].copy()\n    matching_rows['variable_name'] = variable_name  # Add the additional column indicating the variable_name\n    filtered_rows.append(matching_rows)\n\n# Concatenate all filtered rows into a single DataFrame\nresult_df = pd.concat(filtered_rows)\nresult_df = result_df.reset_index(drop=True)\nresult_df","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:50:55.863385Z","iopub.execute_input":"2023-09-17T14:50:55.863768Z","iopub.status.idle":"2023-09-17T14:50:57.103802Z","shell.execute_reply.started":"2023-09-17T14:50:55.863690Z","shell.execute_reply":"2023-09-17T14:50:57.102560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We 'paste' in the new results into the sample submission file\nfor i,col in enumerate( submit_df.columns ):\n    if col == 'id':\n        continue\n    submit_df[col] = result_df[col]\n    if (i%1000) == 0: print(i,col)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:50:57.105337Z","iopub.execute_input":"2023-09-17T14:50:57.105706Z","iopub.status.idle":"2023-09-17T14:51:05.910590Z","shell.execute_reply.started":"2023-09-17T14:50:57.105675Z","shell.execute_reply":"2023-09-17T14:51:05.909420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Quickly check our submission - make sure it matches the given format!\n\ndisplay(submit_df)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:51:05.912060Z","iopub.execute_input":"2023-09-17T14:51:05.912475Z","iopub.status.idle":"2023-09-17T14:51:06.024481Z","shell.execute_reply.started":"2023-09-17T14:51:05.912436Z","shell.execute_reply":"2023-09-17T14:51:06.023388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Add cell type specific transcriptional signatures","metadata":{}},{"cell_type":"code","source":"#group by cell type and take the mean\ndf_simple = df_train.iloc[:, [0] + list(range(5, df_train.shape[1]))]\n\ncell_type_mean = df_simple.groupby('cell_type').mean().reset_index()\n\ncell_type_mean","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:51:06.025798Z","iopub.execute_input":"2023-09-17T14:51:06.026658Z","iopub.status.idle":"2023-09-17T14:51:06.219020Z","shell.execute_reply.started":"2023-09-17T14:51:06.026615Z","shell.execute_reply":"2023-09-17T14:51:06.218199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# List to store the filtered rows\nfiltered_rows = []\n\nfor variable_name in df_ids['cell_type']:\n    matching_rows = cell_type_mean[cell_type_mean['cell_type'] == variable_name].copy()\n    matching_rows['variable_name'] = variable_name  # Add the additional column indicating the variable_name\n    filtered_rows.append(matching_rows)\n\n# Concatenate all filtered rows into a single DataFrame\nresult_df = pd.concat(filtered_rows)\nresult_df = result_df.reset_index(drop=True)\nresult_df","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:51:06.220397Z","iopub.execute_input":"2023-09-17T14:51:06.221255Z","iopub.status.idle":"2023-09-17T14:51:07.400225Z","shell.execute_reply.started":"2023-09-17T14:51:06.221224Z","shell.execute_reply":"2023-09-17T14:51:07.399107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load in the sample template \ntemp_path = '/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'\nsubmit_df_celltype = pd.read_csv(temp_path)\n\n# We 'paste' in the new results into the sample submission file\nfor i,col in enumerate( submit_df_celltype.columns ):\n    if col == 'id':\n        continue\n    submit_df_celltype[col] = result_df[col]\n    if (i%1000) == 0: print(i,col)\n        \nsubmit_df_celltype","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:51:07.401805Z","iopub.execute_input":"2023-09-17T14:51:07.402237Z","iopub.status.idle":"2023-09-17T14:51:21.741810Z","shell.execute_reply.started":"2023-09-17T14:51:07.402197Z","shell.execute_reply":"2023-09-17T14:51:21.740615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Simple ensemble","metadata":{}},{"cell_type":"code","source":"# Lets now combine our submission file with a high scoring one by Alex (Lb = 0.623)\n# source : Version 18 of https://www.kaggle.com/code/alexandervc/op2-eda-baseline-s \n\nfn = '/kaggle/input/alex-v18-denoise-output/submission.csv'\nv18sub = pd.read_csv(fn)\nprint(v18sub.shape)\n\n# Take the average\naverage_df = v18sub*0.45 + submit_df*0.45 + submit_df_celltype*0.1 \n\n# convert 'id' back to int32 or we get an error \naverage_df['id'] = average_df['id'].astype('int32')\n\n# make csv file\naverage_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-09-17T14:51:21.744692Z","iopub.execute_input":"2023-09-17T14:51:21.745036Z","iopub.status.idle":"2023-09-17T14:52:18.515982Z","shell.execute_reply.started":"2023-09-17T14:51:21.745002Z","shell.execute_reply":"2023-09-17T14:52:18.515022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Done!\n\nFeel free to leave comments and suggestions on how to improve!","metadata":{}}]}