{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30559,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import time\nt0start = time.time()\nfrom fastai.collab import *\nfrom fastai.tabular.all import *","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-04T17:57:20.741377Z","iopub.execute_input":"2023-12-04T17:57:20.742518Z","iopub.status.idle":"2023-12-04T17:57:24.777855Z","shell.execute_reply.started":"2023-12-04T17:57:20.742457Z","shell.execute_reply":"2023-12-04T17:57:24.776981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Set Random Seeds","metadata":{}},{"cell_type":"code","source":"def random_seed(seed_value, use_cuda):\n    np.random.seed(seed_value) # for numpy random\n    torch.manual_seed(seed_value) # for pytorch\n    random.seed(seed_value) # for python random\n    if use_cuda:\n        torch.cuda.manual_seed(seed_value)\n        torch.cuda.manual_seed_all(seed_value)\n        torch.backends.cudnn.deterministic = True\n        torch.backends.cudnn.benchmark = False\n        \nseed_value=42\nrandom_seed(seed_value, use_cuda=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:24.779849Z","iopub.execute_input":"2023-12-04T17:57:24.780615Z","iopub.status.idle":"2023-12-04T17:57:24.791501Z","shell.execute_reply.started":"2023-12-04T17:57:24.780579Z","shell.execute_reply":"2023-12-04T17:57:24.790367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read Data","metadata":{}},{"cell_type":"markdown","source":"Here I read the training and test data and melt it to yield a ```DataFrame``` with three categorical features (```cell_type```, ```sm_name```, and ```gene```) and one target (```value```).","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_de_train = pd.read_parquet(fn)\nprint(df_de_train.shape)\ndf_de_train","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:24.792815Z","iopub.execute_input":"2023-12-04T17:57:24.793080Z","iopub.status.idle":"2023-12-04T17:57:27.275469Z","shell.execute_reply.started":"2023-12-04T17:57:24.793056Z","shell.execute_reply":"2023-12-04T17:57:27.274524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_melted = df_de_train.melt(id_vars=['cell_type', 'sm_name'], value_vars=df_de_train.iloc[:,5:].columns, var_name='gene', value_name='value')\ntrain_df_melted","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:27.278126Z","iopub.execute_input":"2023-12-04T17:57:27.278406Z","iopub.status.idle":"2023-12-04T17:57:29.578925Z","shell.execute_reply.started":"2023-12-04T17:57:27.278381Z","shell.execute_reply":"2023-12-04T17:57:29.577977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_id_map = pd.read_csv(fn, index_col=0)\ncols_to_add = df_de_train.iloc[:,5:].columns\ndf_zeros = pd.DataFrame(0.0, columns=cols_to_add, index=df_id_map.index)\ntest_df = pd.concat([df_id_map, df_zeros], axis=1)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:29.580319Z","iopub.execute_input":"2023-12-04T17:57:29.580714Z","iopub.status.idle":"2023-12-04T17:57:29.693466Z","shell.execute_reply.started":"2023-12-04T17:57:29.580673Z","shell.execute_reply":"2023-12-04T17:57:29.692457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_melted = test_df.melt(id_vars=['cell_type', 'sm_name'], value_vars=test_df.iloc[:,2:].columns, var_name='gene', value_name='value')\ntest_df_melted","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:29.694727Z","iopub.execute_input":"2023-12-04T17:57:29.695027Z","iopub.status.idle":"2023-12-04T17:57:31.506368Z","shell.execute_reply.started":"2023-12-04T17:57:29.695002Z","shell.execute_reply":"2023-12-04T17:57:31.505274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define Functions","metadata":{}},{"cell_type":"markdown","source":"The function ```get_denoised_data``` takes the original training data, as well as the melted ```DataFrame``` and returns a melted ```DataFrame``` that has been denoised using PCA and ```n_comp``` components. It also randomly selects and marks a fraction of ```0.2``` of the training set as validation data and replaces that data with the original non-denoised data.","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\n\ndef get_denoised_data(data, data_melted, n_comp=35):\n    Y = data.iloc[:,5:]\n\n    # Standardize the data\n    scaler = StandardScaler()\n    Y_std = scaler.fit_transform(Y)\n\n    reducer = PCA(n_components=n_comp, random_state=seed_value)\n\n    Yr_std = reducer.fit_transform(Y_std)\n    Yr_inv_trans_sdt = reducer.inverse_transform(Yr_std)\n    Yr_inv_trans_denorm = scaler.inverse_transform(Yr_inv_trans_sdt)\n    df_red_inv_trans = pd.DataFrame(Yr_inv_trans_denorm, columns = data.columns[5:])\n    df_de_train_denoised = pd.concat([data.iloc[:, :2], df_red_inv_trans], axis=1)\n    train_denoised_melted = df_de_train_denoised.melt(id_vars=['cell_type', 'sm_name'], value_vars=df_de_train_denoised.iloc[:,2:].columns, var_name='gene', value_name='value')\n    train_denoised_melted['is_valid'] = False\n    valid_indices = data_melted.sample(frac=0.2, random_state=seed_value).index.sort_values().tolist()\n    train_denoised_melted.loc[valid_indices, 'value'] = data_melted.loc[valid_indices, 'value']\n    train_denoised_melted.loc[valid_indices, 'is_valid'] = True\n    return train_denoised_melted","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:31.507708Z","iopub.execute_input":"2023-12-04T17:57:31.508105Z","iopub.status.idle":"2023-12-04T17:57:31.650122Z","shell.execute_reply.started":"2023-12-04T17:57:31.508072Z","shell.execute_reply":"2023-12-04T17:57:31.649247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This splitter function is fed into fast.ai's ```TabularPandas``` to get the test-valid-split.","metadata":{}},{"cell_type":"code","source":"def splitter(df):\n    train = df.index[~df['is_valid']].tolist()\n    valid = df.index[df['is_valid']].tolist()\n    return L(train), L(valid)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:31.651657Z","iopub.execute_input":"2023-12-04T17:57:31.652344Z","iopub.status.idle":"2023-12-04T17:57:31.657454Z","shell.execute_reply.started":"2023-12-04T17:57:31.652307Z","shell.execute_reply":"2023-12-04T17:57:31.656533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The function ```get_dls``` takes the original training data and the melted ```DataFrame```, applies ```get_denoised_data```, calls fast.ai's ```TabularPandas``` to get the training data into the correct format and returns the batched data loaders to feed to the model.","metadata":{}},{"cell_type":"code","source":"def get_dls(data, data_melted, n_comp=35, bs=4096):\n    train_denoised_melted = get_denoised_data(data, data_melted, n_comp=n_comp)\n    cats = train_denoised_melted.iloc[:, :3].columns.to_list()\n    splits = splitter(train_denoised_melted)\n    to = TabularPandas(train_denoised_melted, procs=[Categorify], cat_names=cats, y_names='value', splits=splits)\n    dls = to.dataloaders(bs)\n    return dls","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:31.659953Z","iopub.execute_input":"2023-12-04T17:57:31.660302Z","iopub.status.idle":"2023-12-04T17:57:31.669421Z","shell.execute_reply.started":"2023-12-04T17:57:31.660269Z","shell.execute_reply":"2023-12-04T17:57:31.668527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rmse(preds, targs):\n    return ((targs-preds)**2).mean().sqrt().item()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:31.673266Z","iopub.execute_input":"2023-12-04T17:57:31.674169Z","iopub.status.idle":"2023-12-04T17:57:31.682298Z","shell.execute_reply.started":"2023-12-04T17:57:31.674135Z","shell.execute_reply":"2023-12-04T17:57:31.681447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train Tabular Learner with PCA Denoising Using 10 Components","metadata":{}},{"cell_type":"code","source":"%%time\ndls = get_dls(df_de_train, train_df_melted, n_comp=10, bs=4096)\ndls.valid.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:31.683653Z","iopub.execute_input":"2023-12-04T17:57:31.684377Z","iopub.status.idle":"2023-12-04T17:57:45.964280Z","shell.execute_reply.started":"2023-12-04T17:57:31.684337Z","shell.execute_reply":"2023-12-04T17:57:45.963327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.train.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:45.965625Z","iopub.execute_input":"2023-12-04T17:57:45.966288Z","iopub.status.idle":"2023-12-04T17:57:46.052643Z","shell.execute_reply.started":"2023-12-04T17:57:45.966258Z","shell.execute_reply":"2023-12-04T17:57:46.051760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here I define an ```emb_szs``` dictionary to change the default dimension of ```gene``` embeddings from ```389``` to ```600```:","metadata":{}},{"cell_type":"code","source":"emb_szs = {'cell_type': 5, 'sm_name': 26, 'gene': 600}\nget_emb_sz(dls.train_ds)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:46.053797Z","iopub.execute_input":"2023-12-04T17:57:46.054064Z","iopub.status.idle":"2023-12-04T17:57:46.060702Z","shell.execute_reply.started":"2023-12-04T17:57:46.054039Z","shell.execute_reply":"2023-12-04T17:57:46.059560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The next code block is to find the min and max values of the target, because the ```tabular_learner``` tends to learn better if the target range (```y_range```) is provided.","metadata":{}},{"cell_type":"code","source":"y = dls.train.y\ny_min = np.ceil(y.min()).astype(int)\ny_max = np.ceil(y.max()).astype(int)\ny_min, y_max","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:46.061941Z","iopub.execute_input":"2023-12-04T17:57:46.062201Z","iopub.status.idle":"2023-12-04T17:57:46.095873Z","shell.execute_reply.started":"2023-12-04T17:57:46.062178Z","shell.execute_reply":"2023-12-04T17:57:46.094861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = tabular_learner(dls, y_range=(y_min, y_max), emb_szs=emb_szs, layers=[1000, 500, 250], n_out=1, loss_func=F.mse_loss)\nlearn.fit_one_cycle(20, 3e-4)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T17:57:46.097335Z","iopub.execute_input":"2023-12-04T17:57:46.097695Z","iopub.status.idle":"2023-12-04T18:18:24.789713Z","shell.execute_reply.started":"2023-12-04T17:57:46.097662Z","shell.execute_reply":"2023-12-04T18:18:24.788660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds, targs = learn.get_preds()\nrmse(preds, targs)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:18:24.791004Z","iopub.execute_input":"2023-12-04T18:18:24.791318Z","iopub.status.idle":"2023-12-04T18:18:31.222423Z","shell.execute_reply.started":"2023-12-04T18:18:24.791292Z","shell.execute_reply":"2023-12-04T18:18:31.221402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"test_dl = dls.test_dl(test_df_melted)\npreds, _ = learn.get_preds(dl=test_dl)\npreds","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:18:31.223877Z","iopub.execute_input":"2023-12-04T18:18:31.224498Z","iopub.status.idle":"2023-12-04T18:18:46.388139Z","shell.execute_reply.started":"2023-12-04T18:18:31.224445Z","shell.execute_reply":"2023-12-04T18:18:46.387164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the last step I reshape the predictions back into a ```255 x 18211``` tensor for submission:","metadata":{}},{"cell_type":"code","source":"to_submit = preds.view(18211, -1).t().numpy()\nsubmit = pd.DataFrame(to_submit, columns=df_de_train.iloc[:,5:].columns)\nsubmit.index.name = 'id'\nsubmit","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:18:46.389359Z","iopub.execute_input":"2023-12-04T18:18:46.389693Z","iopub.status.idle":"2023-12-04T18:18:46.454983Z","shell.execute_reply.started":"2023-12-04T18:18:46.389667Z","shell.execute_reply":"2023-12-04T18:18:46.453670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-04T18:18:46.456270Z","iopub.execute_input":"2023-12-04T18:18:46.456623Z","iopub.status.idle":"2023-12-04T18:18:54.360938Z","shell.execute_reply.started":"2023-12-04T18:18:46.456595Z","shell.execute_reply":"2023-12-04T18:18:54.359910Z"},"trusted":true},"execution_count":null,"outputs":[]}]}