{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30616,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Please Read","metadata":{}},{"cell_type":"markdown","source":"The following notebook is an adaptation of a popular [notebook](https://www.kaggle.com/code/jhoward/getting-started-with-nlp-for-absolute-beginners) by Jeremy Howard discussed in fast.ai's [Practical Deep Learning for Coders – Lesson 4](https://course.fast.ai/Lessons/lesson4.html).\n\nIn order for this notebook to run on Kaggle in a reasonable amount of time (~ 2 h) the model will be trained on only 2 % of the training data sampled randomly. Please note that this notebook serves only as a training demo, since for best results the model needs to be fine-tuned on all training data. For context: Fine-tuning the model for 5 epochs on an A10 GPU using a batch size of 256 and a train-valid-split of 80/20 took about 12 hours. For predictions submitted to the Open Problems – Single-Cell Perturbation competition the model was fine-tuned for 20–25 epochs.","metadata":{}},{"cell_type":"markdown","source":"## Imports and Random Seed","metadata":{}},{"cell_type":"code","source":"import time\nt0start = time.time()\nfrom fastai.collab import *\nfrom fastai.tabular.all import *\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:20:35.514861Z","iopub.execute_input":"2023-12-09T02:20:35.515561Z","iopub.status.idle":"2023-12-09T02:20:41.660190Z","shell.execute_reply.started":"2023-12-09T02:20:35.515531Z","shell.execute_reply":"2023-12-09T02:20:41.659256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_seed = 42","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:20:41.662130Z","iopub.execute_input":"2023-12-09T02:20:41.663015Z","iopub.status.idle":"2023-12-09T02:20:41.667298Z","shell.execute_reply.started":"2023-12-09T02:20:41.662976Z","shell.execute_reply":"2023-12-09T02:20:41.666293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import set_seed\nset_seed(random_seed)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:20:41.668733Z","iopub.execute_input":"2023-12-09T02:20:41.669041Z","iopub.status.idle":"2023-12-09T02:20:53.901244Z","shell.execute_reply.started":"2023-12-09T02:20:41.669008Z","shell.execute_reply":"2023-12-09T02:20:53.900393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Train Set","metadata":{}},{"cell_type":"markdown","source":"Here I read the training data and melt it to yield a ```DataFrame``` with three categorical features (```cell_type```, ```sm_name```, and ```gene```) and one target (```value```).","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_de_train = pd.read_parquet(fn)# , index_col = 0)\ntrain_df = df_de_train.drop(columns=['sm_lincs_id', 'SMILES', 'control'])\ntrdf = train_df.melt(id_vars=['cell_type', 'sm_name'], value_vars=train_df.iloc[:,2:].columns, var_name='gene', value_name='value')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:20:53.903223Z","iopub.execute_input":"2023-12-09T02:20:53.905003Z","iopub.status.idle":"2023-12-09T02:20:58.712751Z","shell.execute_reply.started":"2023-12-09T02:20:53.904972Z","shell.execute_reply":"2023-12-09T02:20:58.711773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The next line samples 2 % of the training data to demonstrate model training in a reasonable time frame:","metadata":{}},{"cell_type":"code","source":"trdf = trdf.sample(frac=0.02).reset_index(drop=true)","metadata":{"execution":{"iopub.status.busy":"2023-12-09T02:20:58.714050Z","iopub.execute_input":"2023-12-09T02:20:58.714349Z","iopub.status.idle":"2023-12-09T02:20:59.436310Z","shell.execute_reply.started":"2023-12-09T02:20:58.714321Z","shell.execute_reply":"2023-12-09T02:20:59.435247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Test Set","metadata":{}},{"cell_type":"markdown","source":"Similar procedure for the test set.","metadata":{}},{"cell_type":"code","source":"fn = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_id_map = pd.read_csv(fn)\nfn = '/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'\ndf = pd.read_csv(fn, index_col = 0)\n\ncols_to_add = df_de_train.iloc[:,5:].columns\ncols_to_add\n\ndf_zeros = pd.DataFrame(0.0, columns=cols_to_add, index=df_id_map.index)\ndf_zeros\n\ndf_id_map_preds = pd.concat([df_id_map, df_zeros], axis=1)\ntsdf = df_id_map_preds.melt(id_vars=['cell_type', 'sm_name'], value_vars=df_id_map_preds.iloc[:,3:].columns, var_name='gene', value_name='value')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:20:59.437711Z","iopub.execute_input":"2023-12-09T02:20:59.438014Z","iopub.status.idle":"2023-12-09T02:21:04.370808Z","shell.execute_reply.started":"2023-12-09T02:20:59.437987Z","shell.execute_reply":"2023-12-09T02:21:04.369926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inspect Train and Test Set","metadata":{}},{"cell_type":"code","source":"trdf","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:21:04.371977Z","iopub.execute_input":"2023-12-09T02:21:04.372290Z","iopub.status.idle":"2023-12-09T02:21:04.391844Z","shell.execute_reply.started":"2023-12-09T02:21:04.372261Z","shell.execute_reply":"2023-12-09T02:21:04.390790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tsdf","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:21:04.392967Z","iopub.execute_input":"2023-12-09T02:21:04.393260Z","iopub.status.idle":"2023-12-09T02:21:04.408626Z","shell.execute_reply.started":"2023-12-09T02:21:04.393233Z","shell.execute_reply":"2023-12-09T02:21:04.407527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Transformer Input","metadata":{}},{"cell_type":"code","source":"trdf['input'] = 'CELL TYPE: ' + trdf.cell_type + '; TREATMENT: ' + trdf.sm_name + '; GENE: ' + trdf.gene\ntrdf.input.head()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:21:04.410301Z","iopub.execute_input":"2023-12-09T02:21:04.410739Z","iopub.status.idle":"2023-12-09T02:21:04.571160Z","shell.execute_reply.started":"2023-12-09T02:21:04.410700Z","shell.execute_reply":"2023-12-09T02:21:04.570131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tsdf['input'] = 'CELL TYPE: ' + tsdf.cell_type + '; TREATMENT: ' + tsdf.sm_name + '; GENE: ' + tsdf.gene\ntsdf.input.head()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:21:04.574890Z","iopub.execute_input":"2023-12-09T02:21:04.575207Z","iopub.status.idle":"2023-12-09T02:21:08.430476Z","shell.execute_reply.started":"2023-12-09T02:21:04.575178Z","shell.execute_reply":"2023-12-09T02:21:08.429430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datasets import Dataset, DatasetDict\ntrds = Dataset.from_pandas(trdf)\ntrds","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:21:08.431464Z","iopub.execute_input":"2023-12-09T02:21:08.431729Z","iopub.status.idle":"2023-12-09T02:21:08.818758Z","shell.execute_reply.started":"2023-12-09T02:21:08.431705Z","shell.execute_reply":"2023-12-09T02:21:08.817728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tsds = Dataset.from_pandas(tsdf)\ntsds","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:21:08.820039Z","iopub.execute_input":"2023-12-09T02:21:08.820791Z","iopub.status.idle":"2023-12-09T02:21:10.526371Z","shell.execute_reply.started":"2023-12-09T02:21:08.820751Z","shell.execute_reply":"2023-12-09T02:21:10.525468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Model","metadata":{}},{"cell_type":"markdown","source":"### Tokenize","metadata":{}},{"cell_type":"code","source":"from transformers import AutoModelForSequenceClassification, AutoTokenizer\nmodel_nm = 'microsoft/deberta-v3-small'\ntokz = AutoTokenizer.from_pretrained(model_nm)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:21:10.527593Z","iopub.execute_input":"2023-12-09T02:21:10.527920Z","iopub.status.idle":"2023-12-09T02:21:13.335839Z","shell.execute_reply.started":"2023-12-09T02:21:10.527890Z","shell.execute_reply":"2023-12-09T02:21:13.334942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"L(tokz.tokenize(trds['input'][0]))","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:21:13.337015Z","iopub.execute_input":"2023-12-09T02:21:13.337331Z","iopub.status.idle":"2023-12-09T02:21:13.669900Z","shell.execute_reply.started":"2023-12-09T02:21:13.337302Z","shell.execute_reply":"2023-12-09T02:21:13.668835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tok_func(x): return tokz(x[\"input\"])\n\ntok_trds = trds.map(tok_func, batched = True)\ntok_tsds = tsds.map(tok_func, batched = True)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:21:13.671379Z","iopub.execute_input":"2023-12-09T02:21:13.671700Z","iopub.status.idle":"2023-12-09T02:25:59.844046Z","shell.execute_reply.started":"2023-12-09T02:21:13.671672Z","shell.execute_reply":"2023-12-09T02:25:59.842278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tok_trds.save_to_disk('tokenized-train-dataset')\ntok_tsds.save_to_disk('tokenized-test-dataset')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:25:59.846145Z","iopub.execute_input":"2023-12-09T02:25:59.846502Z","iopub.status.idle":"2023-12-09T02:26:01.526410Z","shell.execute_reply.started":"2023-12-09T02:25:59.846456Z","shell.execute_reply":"2023-12-09T02:26:01.525649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Started tokenization...')\nfrom datasets import load_from_disk\ntok_trds = load_from_disk('tokenized-train-dataset')\ntok_tsds = load_from_disk('tokenized-test-dataset')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:26:01.527551Z","iopub.execute_input":"2023-12-09T02:26:01.527837Z","iopub.status.idle":"2023-12-09T02:26:02.681642Z","shell.execute_reply.started":"2023-12-09T02:26:01.527811Z","shell.execute_reply":"2023-12-09T02:26:02.680717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tok_trds[0]['input'], L(tok_trds[0]['input_ids'])","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:26:02.684209Z","iopub.execute_input":"2023-12-09T02:26:02.684546Z","iopub.status.idle":"2023-12-09T02:26:02.692281Z","shell.execute_reply.started":"2023-12-09T02:26:02.684519Z","shell.execute_reply":"2023-12-09T02:26:02.691331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tok_tsds[0]['input'], L(tok_tsds[0]['input_ids'])","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:26:02.693435Z","iopub.execute_input":"2023-12-09T02:26:02.693745Z","iopub.status.idle":"2023-12-09T02:26:02.704480Z","shell.execute_reply.started":"2023-12-09T02:26:02.693719Z","shell.execute_reply":"2023-12-09T02:26:02.703500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in tok_tsds[0]['input_ids']:\n    print(tokz.decode(i))","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:26:02.705666Z","iopub.execute_input":"2023-12-09T02:26:02.705992Z","iopub.status.idle":"2023-12-09T02:26:02.714629Z","shell.execute_reply.started":"2023-12-09T02:26:02.705958Z","shell.execute_reply":"2023-12-09T02:26:02.713751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tok_trds = tok_trds.rename_columns({\"value\":\"labels\"})\ntok_trds","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:26:02.715765Z","iopub.execute_input":"2023-12-09T02:26:02.716034Z","iopub.status.idle":"2023-12-09T02:26:02.732572Z","shell.execute_reply.started":"2023-12-09T02:26:02.716010Z","shell.execute_reply":"2023-12-09T02:26:02.731697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Split Train Data","metadata":{}},{"cell_type":"code","source":"dds = tok_trds.train_test_split(0.20, seed=random_seed)\ndds","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:26:02.733604Z","iopub.execute_input":"2023-12-09T02:26:02.733871Z","iopub.status.idle":"2023-12-09T02:26:02.832907Z","shell.execute_reply.started":"2023-12-09T02:26:02.733846Z","shell.execute_reply":"2023-12-09T02:26:02.831939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Define Model","metadata":{}},{"cell_type":"code","source":"bs = 64\nepochs = 5\nlr = 8e-5","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:29:17.845104Z","iopub.execute_input":"2023-12-09T02:29:17.845565Z","iopub.status.idle":"2023-12-09T02:29:17.850360Z","shell.execute_reply.started":"2023-12-09T02:29:17.845530Z","shell.execute_reply":"2023-12-09T02:29:17.849145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"steps = len(dds['train']) // bs\nsteps","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:29:18.202940Z","iopub.execute_input":"2023-12-09T02:29:18.204012Z","iopub.status.idle":"2023-12-09T02:29:18.211522Z","shell.execute_reply.started":"2023-12-09T02:29:18.203965Z","shell.execute_reply":"2023-12-09T02:29:18.210292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import TrainingArguments, Trainer\n\nargs = TrainingArguments('outputs', save_steps=steps, learning_rate=lr, warmup_ratio=0.1, lr_scheduler_type='cosine', fp16=True,\n    evaluation_strategy=\"epoch\", per_device_train_batch_size=bs, per_device_eval_batch_size=bs*2,\n    num_train_epochs=epochs, weight_decay=0.01, report_to='none', seed=random_seed) # resume_from_checkpoint=\"/kaggle/working/20231005/outputs/checkpoint-104826\"","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:29:18.445493Z","iopub.execute_input":"2023-12-09T02:29:18.446407Z","iopub.status.idle":"2023-12-09T02:29:18.456281Z","shell.execute_reply.started":"2023-12-09T02:29:18.446357Z","shell.execute_reply":"2023-12-09T02:29:18.455048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = AutoModelForSequenceClassification.from_pretrained(model_nm, num_labels=1) # for checkpoint use e.g. \"/kaggle/working/20231005/outputs/checkpoint-104826\" instead of model_nm\ntrainer = Trainer(model, args, train_dataset=dds['train'], eval_dataset=dds['test'],\n                  tokenizer=tokz)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:29:18.815328Z","iopub.execute_input":"2023-12-09T02:29:18.816156Z","iopub.status.idle":"2023-12-09T02:29:20.548406Z","shell.execute_reply.started":"2023-12-09T02:29:18.816117Z","shell.execute_reply":"2023-12-09T02:29:20.547364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"print('Started training...')\ntrainer.train()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-09T02:29:20.550107Z","iopub.execute_input":"2023-12-09T02:29:20.550415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_string = 'deberta-v3-small_sparse_prompt_5eps42'\ntrainer.save_model(f'./{model_string}/')","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_loss_history = [log['eval_loss'] for log in trainer.state.log_history if 'eval_loss' in log]\nvalid_step_history = [log['step'] for log in trainer.state.log_history if 'eval_loss' in log]\ntrain_loss_history = [log['loss'] for log in trainer.state.log_history if 'loss' in log]\ntrain_step_history = [log['step'] for log in trainer.state.log_history if 'loss' in log]\nlrate_history = [log['learning_rate'] for log in trainer.state.log_history if 'loss' in log]\nepoch_history = [log['epoch'] for log in trainer.state.log_history if 'eval_loss' in log]\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 5))\n\nax1.plot(valid_step_history, valid_loss_history)\nax1.plot(train_step_history, train_loss_history)\nax1.set_xlabel('Step')\nax1.set_ylabel('Loss')\nax1.set_title('Loss History')\n\nax2.plot(train_step_history, lrate_history)\nax2.set_xlabel('Step')\nax2.set_ylabel('Learning Rate')\nax2.set_title('Learning Rate History')\n\nplt.savefig(f'{model_string}_training.pdf', format='pdf')\nplt.tight_layout\nplt.show()","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(f'{model_string}_training_history.txt', 'w') as file:\n    file.write(str(trainer.state.log_history))","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predict","metadata":{}},{"cell_type":"code","source":"print('Started inference...')\npreds = trainer.predict(tok_tsds).predictions.astype(float)\npreds","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.max(), preds.min()","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_rounded = np.round_(preds, 4)\npreds_rounded","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the last step I reshape the predictions back into a 255 x 18211 tensor for submission:","metadata":{}},{"cell_type":"code","source":"to_submit = preds_rounded.reshape(18211, -1).T","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.DataFrame(to_submit, columns=df_de_train.iloc[:,5:].columns)\nsubmit.index.name = 'id'\nsubmit","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv(f'submission.csv')","metadata":{"tags":[],"trusted":true},"execution_count":null,"outputs":[]}]}