{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":7712331,"sourceType":"datasetVersion","datasetId":4441094}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Preparing environment","metadata":{}},{"cell_type":"code","source":"WORKING_DIR = '/kaggle/working'\nINPUT_DIR = '/kaggle/input/open-problems-single-cell-perturbations'","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:29:05.342574Z","iopub.execute_input":"2024-04-08T10:29:05.343320Z","iopub.status.idle":"2024-04-08T10:29:05.348000Z","shell.execute_reply.started":"2024-04-08T10:29:05.343277Z","shell.execute_reply":"2024-04-08T10:29:05.347006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir $WORKING_DIR/data\n!mkdir $WORKING_DIR/model","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:07:17.023004Z","iopub.execute_input":"2024-04-08T10:07:17.023319Z","iopub.status.idle":"2024-04-08T10:07:18.892540Z","shell.execute_reply.started":"2024-04-08T10:07:17.023294Z","shell.execute_reply":"2024-04-08T10:07:18.891287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://github.com/lzj1769/7th_place_solution_Single-Cell-Perturbations.git $WORKING_DIR/7th_place_solution_Single-Cell-Perturbations","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:08:14.180934Z","iopub.execute_input":"2024-04-08T10:08:14.181273Z","iopub.status.idle":"2024-04-08T10:08:15.947325Z","shell.execute_reply.started":"2024-04-08T10:08:14.181244Z","shell.execute_reply":"2024-04-08T10:08:15.946407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd $WORKING_DIR/7th_place_solution_Single-Cell-Perturbations/src","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:08:15.949074Z","iopub.execute_input":"2024-04-08T10:08:15.949380Z","iopub.status.idle":"2024-04-08T10:08:15.955810Z","shell.execute_reply.started":"2024-04-08T10:08:15.949349Z","shell.execute_reply":"2024-04-08T10:08:15.954955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.append(f'{WORKING_DIR}/7th_place_solution_Single-Cell-Perturbations/src/')","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:08:56.615579Z","iopub.execute_input":"2024-04-08T10:08:56.616575Z","iopub.status.idle":"2024-04-08T10:08:56.620569Z","shell.execute_reply.started":"2024-04-08T10:08:56.616531Z","shell.execute_reply":"2024-04-08T10:08:56.619675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Just for testing (make dataset smaller)","metadata":{}},{"cell_type":"markdown","source":"This section just for testing full pipeline without using large resouses. It cuts the amount of genes from 18215 to 95. Becouse of this the **final result not applicable to submission!**","metadata":{}},{"cell_type":"code","source":"OLD_INPUT_DIR = INPUT_DIR\nINPUT_DIR = WORKING_DIR + '/input'","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:29:30.368841Z","iopub.execute_input":"2024-04-08T10:29:30.369574Z","iopub.status.idle":"2024-04-08T10:29:30.373628Z","shell.execute_reply.started":"2024-04-08T10:29:30.369539Z","shell.execute_reply":"2024-04-08T10:29:30.372674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir $INPUT_DIR\n!cp -r $OLD_INPUT_DIR/. $INPUT_DIR/.","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:09:29.192781Z","iopub.execute_input":"2024-04-08T10:09:29.193131Z","iopub.status.idle":"2024-04-08T10:10:33.402575Z","shell.execute_reply.started":"2024-04-08T10:09:29.193103Z","shell.execute_reply":"2024-04-08T10:10:33.401442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport os\nimport pandas as pd\n\nGENE_NUMBER = 95\nin_path = re.sub(r\"\\\\\", \"\", f\"{OLD_INPUT_DIR}\")\nout_path = re.sub(r\"\\\\\", \"\", f\"{INPUT_DIR}\")\n\ndf_train = pd.read_parquet(os.path.join(in_path, 'de_train.parquet'))\ndf_test = pd.read_csv(os.path.join(in_path, 'sample_submission.csv'))\n\nprint('source train shape: ', df_train.shape)\nprint('source test shape: ', df_test.shape)\n\ndf_train = df_train.drop(columns=df_train.columns[GENE_NUMBER+4:])\ndf_train.to_parquet(os.path.join(out_path, 'de_train.parquet'))\n\n\ndf_test = df_test.drop(columns=df_test.columns[GENE_NUMBER:])\ndf_test.to_csv(os.path.join(out_path, 'sample_submission.csv'), encoding='utf-8', index=False)\n\nprint('new train shape: ', df_train.shape)\nprint('new test shape: ', df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:29:33.339101Z","iopub.execute_input":"2024-04-08T10:29:33.340035Z","iopub.status.idle":"2024-04-08T10:29:36.752750Z","shell.execute_reply.started":"2024-04-08T10:29:33.339996Z","shell.execute_reply":"2024-04-08T10:29:36.751864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing data","metadata":{}},{"cell_type":"markdown","source":"This part converting dataframe to long format","metadata":{}},{"cell_type":"code","source":"!python 01_prepare_data.py \\\n    --de_train $INPUT_DIR/de_train.parquet \\\n    --output_filename $WORKING_DIR/data/train.csv","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:10:38.588693Z","iopub.execute_input":"2024-04-08T10:10:38.588995Z","iopub.status.idle":"2024-04-08T10:10:40.569192Z","shell.execute_reply.started":"2024-04-08T10:10:38.588969Z","shell.execute_reply":"2024-04-08T10:10:40.567984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n\n# def convert_data(de_train_file, output_file):\n#     df = pd.read_parquet(de_train_file)\n\n#     df = df.sort_values([\"cell_type\", \"sm_name\"])\n#     df = df.drop([\"sm_lincs_id\", \"SMILES\", \"control\"], axis=1)\n#     df = pd.melt(\n#         df, id_vars=[\"cell_type\", \"sm_name\"], var_name=\"gene\", value_name=\"target\"\n#     )\n#     df.to_csv(output_file)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T07:17:26.728466Z","iopub.execute_input":"2024-03-27T07:17:26.728760Z","iopub.status.idle":"2024-03-27T07:17:26.734197Z","shell.execute_reply.started":"2024-03-27T07:17:26.728732Z","shell.execute_reply":"2024-03-27T07:17:26.733145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# de_train_file = '../input/open-problems-single-cell-perturbations/de_train.parquet'\n# output_file = 'data/train.csv'\n# convert_data(de_train_file, output_file)","metadata":{"execution":{"iopub.status.busy":"2024-03-27T07:17:26.735703Z","iopub.execute_input":"2024-03-27T07:17:26.736052Z","iopub.status.idle":"2024-03-27T07:17:26.745447Z","shell.execute_reply.started":"2024-03-27T07:17:26.735999Z","shell.execute_reply":"2024-03-27T07:17:26.744530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training embedding model","metadata":{}},{"cell_type":"markdown","source":"In this step, the goal is to learn a specific embedding for each cell type, molecular, and gene. \n\n\nApproach is based on the [paper](https://genomebiology.biomedcentral.com/articles/10.1186/s13059-020-01977-6), where the authors used deep tensor factorization to learn a dense, information-rich representation for cell type, experimental assay, and genomic position.\n\nApproach is the same, but the model architectures and parameters, such as the number of latent factors, and the number of dimensions of the network, as well as how to combine the features (e.g., concatenate vs. additive), were changed by author of Top7 solution.","metadata":{}},{"cell_type":"markdown","source":"The final model looks like: \n\n```\nclass DeepTensorFactorization(torch.nn.Module):\n    def __init__(self, \n                 cell_types, \n                 compounds, \n                 genes, \n                 n_cell_type_factors: int=4, \n                 n_compounds_factors: int=16, \n                 n_gene_factors: int=128,\n                 n_hiddens: int=2048,\n                 dropout: float=0.1):\n        super().__init__()\n\n        self.cell_types = cell_types\n        self.compounds = compounds\n        self.genes = genes\n\n        self.n_cell_types = len(cell_types)\n        self.n_compounds = len(compounds)\n        self.n_genes = len(genes)\n\n        self.n_cell_type_factors = n_cell_type_factors\n        self.n_compounds_factors = n_compounds_factors\n        self.n_gene_factors = n_gene_factors\n\n        self.cell_type_embedding = torch.nn.Embedding(self.n_cell_types, self.n_cell_type_factors)\n        self.compound_embedding = torch.nn.Embedding(self.n_compounds, self.n_compounds_factors)\n        self.gene_embedding = torch.nn.Embedding(self.n_genes, self.n_gene_factors)\n\n        self.n_hiddens = n_hiddens\n        self.dropout = dropout\n        self.n_factors = n_cell_type_factors + n_compounds_factors + n_gene_factors\n\n        self.model = nn.Sequential(nn.Linear(self.n_factors, self.n_hiddens),\n                                   nn.BatchNorm1d(self.n_hiddens),\n                                   nn.ReLU(),\n                                   nn.Dropout(self.dropout),\n                                   nn.Linear(self.n_hiddens, self.n_hiddens),\n                                   nn.BatchNorm1d(self.n_hiddens),\n                                   nn.ReLU(),\n                                   nn.Dropout(self.dropout),\n                                   nn.Linear(self.n_hiddens, 1))\n\n    def forward(self, cell_type_indices, compound_indices, gene_indices):\n        cell_type_vec = self.cell_type_embedding(cell_type_indices)\n        compound_vec = self.compound_embedding(compound_indices)\n        gene_vec = self.gene_embedding(gene_indices)\n\n        x = torch.concat([cell_type_vec, compound_vec, gene_vec], dim=1)\n        x = self.model(x)\n\n        return x\n\n```","metadata":{}},{"cell_type":"code","source":"!python 02_train_embedding.py \\\n    --de_train $INPUT_DIR/de_train.parquet \\\n    --train_data $WORKING_DIR/data/train.csv \\\n    --batch_size 5000 \\\n    --epochs 2 \\\n    --lr 0.001 \\\n    --seed 0 \\\n    --out_dir $WORKING_DIR/model \\\n    --out_name embedding_model \\\n    --log_dir $WORKING_DIR/model\n","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:10:42.374911Z","iopub.execute_input":"2024-04-08T10:10:42.375386Z","iopub.status.idle":"2024-04-08T10:11:05.749105Z","shell.execute_reply.started":"2024-04-08T10:10:42.375346Z","shell.execute_reply":"2024-04-08T10:11:05.748001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Splitting data","metadata":{}},{"cell_type":"markdown","source":"Preparing data for training perturbation model","metadata":{}},{"cell_type":"code","source":"!python 03_split_data.py \\\n    --de_train $INPUT_DIR/de_train.parquet \\\n    --sample_submission $INPUT_DIR/sample_submission.csv \\\n    --id_map $INPUT_DIR/id_map.csv \\\n    --out_dir $WORKING_DIR/data","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:11:05.751185Z","iopub.execute_input":"2024-04-08T10:11:05.751512Z","iopub.status.idle":"2024-04-08T10:11:08.699819Z","shell.execute_reply.started":"2024-04-08T10:11:05.751479Z","shell.execute_reply":"2024-04-08T10:11:08.698624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extracting features","metadata":{}},{"cell_type":"markdown","source":"This section generates embeddings for the train, validation and test set. Embeddings here is the concatination of output of embedding layers for cell_type, sm_name and gene from DeepTensorFactorization model.","metadata":{}},{"cell_type":"code","source":"!mkdir $WORKING_DIR/data/embeddings","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:11:08.701366Z","iopub.execute_input":"2024-04-08T10:11:08.701755Z","iopub.status.idle":"2024-04-08T10:11:09.646308Z","shell.execute_reply.started":"2024-04-08T10:11:08.701710Z","shell.execute_reply":"2024-04-08T10:11:09.644950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python 04_extract_embedding.py \\\n    --de_train $INPUT_DIR/de_train.parquet \\\n    --model $WORKING_DIR/model/embedding_model_epoch_1.pth \\\n    --splited_data_dir $WORKING_DIR/data \\\n    --out_dir $WORKING_DIR/data/embeddings","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:11:09.648919Z","iopub.execute_input":"2024-04-08T10:11:09.649233Z","iopub.status.idle":"2024-04-08T10:11:31.304917Z","shell.execute_reply.started":"2024-04-08T10:11:09.649202Z","shell.execute_reply":"2024-04-08T10:11:31.303603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training prediction model","metadata":{}},{"cell_type":"markdown","source":"Final decision was based on the ensemble of models training for each of the cell types, so we train it separately.","metadata":{}},{"cell_type":"markdown","source":"Perturbation model based on fully connected network with three layers.\n\n```\n\n        self.model = nn.Sequential(nn.Linear(self.n_input, self.n_hiddens),\n                                   nn.BatchNorm1d(self.n_hiddens),\n                                   nn.ReLU(),\n                                   nn.Dropout(self.dropout),\n                                   nn.Linear(self.n_hiddens, self.n_hiddens),\n                                   nn.BatchNorm1d(self.n_hiddens),\n                                   nn.ReLU(),\n                                   nn.Dropout(self.dropout),\n                                   nn.Linear(self.n_hiddens, 1))\n```","metadata":{}},{"cell_type":"code","source":"!python 05_train_prediction_model.py \\\n    --valid_cell_type nk \\\n    --input_dir $WORKING_DIR/data/embeddings \\\n    --batch_size 100 \\\n    --epochs 10 \\\n    --lr 0.001 \\\n    --seed 0 \\\n    --out_dir $WORKING_DIR/model/ \\\n    --out_name nk \\\n    --log_dir $WORKING_DIR/model/\n","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:21:21.080165Z","iopub.execute_input":"2024-04-08T10:21:21.080950Z","iopub.status.idle":"2024-04-08T10:21:59.269175Z","shell.execute_reply.started":"2024-04-08T10:21:21.080910Z","shell.execute_reply":"2024-04-08T10:21:59.268034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python 05_train_prediction_model.py \\\n    --valid_cell_type t_cd4 \\\n    --input_dir $WORKING_DIR/data/embeddings \\\n    --batch_size 100 \\\n    --epochs 10 \\\n    --lr 0.001 \\\n    --seed 0 \\\n    --out_dir $WORKING_DIR/model/ \\\n    --out_name t_cd4 \\\n    --log_dir $WORKING_DIR/model/","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:21:59.271352Z","iopub.execute_input":"2024-04-08T10:21:59.271683Z","iopub.status.idle":"2024-04-08T10:22:37.488233Z","shell.execute_reply.started":"2024-04-08T10:21:59.271650Z","shell.execute_reply":"2024-04-08T10:22:37.487122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python 05_train_prediction_model.py \\\n    --valid_cell_type t_cd8 \\\n    --input_dir $WORKING_DIR/data/embeddings \\\n    --batch_size 100 \\\n    --epochs 10 \\\n    --lr 0.001 \\\n    --seed 0 \\\n    --out_dir $WORKING_DIR/model/ \\\n    --out_name t_cd8 \\\n    --log_dir $WORKING_DIR/model/","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:22:37.489626Z","iopub.execute_input":"2024-04-08T10:22:37.489910Z","iopub.status.idle":"2024-04-08T10:23:15.974929Z","shell.execute_reply.started":"2024-04-08T10:22:37.489880Z","shell.execute_reply":"2024-04-08T10:23:15.973800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python 05_train_prediction_model.py \\\n    --valid_cell_type t_reg \\\n    --input_dir $WORKING_DIR/data/embeddings \\\n    --batch_size 100 \\\n    --epochs 10 \\\n    --lr 0.001 \\\n    --seed 0 \\\n    --out_dir $WORKING_DIR/model/ \\\n    --out_name t_reg \\\n    --log_dir $WORKING_DIR/model/","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:23:15.977122Z","iopub.execute_input":"2024-04-08T10:23:15.977432Z","iopub.status.idle":"2024-04-08T10:23:54.053101Z","shell.execute_reply.started":"2024-04-08T10:23:15.977400Z","shell.execute_reply":"2024-04-08T10:23:54.052142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting","metadata":{}},{"cell_type":"markdown","source":"The final submission is average prediction of all models. It is in $WORKING_DIR/submission/avg_valid_mrrmse_{}.csv","metadata":{}},{"cell_type":"code","source":"!mkdir $WORKING_DIR/submission","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:23:54.054387Z","iopub.execute_input":"2024-04-08T10:23:54.054683Z","iopub.status.idle":"2024-04-08T10:23:54.999029Z","shell.execute_reply.started":"2024-04-08T10:23:54.054651Z","shell.execute_reply":"2024-04-08T10:23:54.997802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python 06_predict.py \\\n    --input_data_dir $WORKING_DIR/data/embeddings \\\n    --model_dir $WORKING_DIR/model/ \\\n    --test_file $WORKING_DIR/data/test.csv \\\n    --submission_dir $WORKING_DIR/submission \\\n    --seed 0","metadata":{"execution":{"iopub.status.busy":"2024-04-08T10:23:55.000558Z","iopub.execute_input":"2024-04-08T10:23:55.000879Z","iopub.status.idle":"2024-04-08T10:24:02.313054Z","shell.execute_reply.started":"2024-04-08T10:23:55.000848Z","shell.execute_reply":"2024-04-08T10:24:02.311919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}