{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"}],"dockerImageVersionId":30559,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import time\nt0start = time.time()\nfrom fastai.collab import *\nfrom fastai.tabular.all import *","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-24T18:58:40.268208Z","iopub.execute_input":"2023-11-24T18:58:40.268518Z","iopub.status.idle":"2023-11-24T18:58:45.120085Z","shell.execute_reply.started":"2023-11-24T18:58:40.268491Z","shell.execute_reply":"2023-11-24T18:58:45.119024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Set Random Seeds","metadata":{}},{"cell_type":"code","source":"def random_seed(seed_value, use_cuda):\n    np.random.seed(seed_value) # for numpy random\n    torch.manual_seed(seed_value) # for pytorch\n    random.seed(seed_value) # for python random\n    if use_cuda:\n        torch.cuda.manual_seed(seed_value)\n        torch.cuda.manual_seed_all(seed_value)\n        torch.backends.cudnn.deterministic = True\n        torch.backends.cudnn.benchmark = False\n        \nseed_value=670\nrandom_seed(seed_value, use_cuda=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:58:46.565517Z","iopub.execute_input":"2023-11-24T18:58:46.566280Z","iopub.status.idle":"2023-11-24T18:58:46.577676Z","shell.execute_reply.started":"2023-11-24T18:58:46.566248Z","shell.execute_reply":"2023-11-24T18:58:46.576629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read Data","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_de_train = pd.read_parquet(fn)# , index_col = 0)\nprint(df_de_train.shape)\ndf_de_train","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:58:48.408323Z","iopub.execute_input":"2023-11-24T18:58:48.409035Z","iopub.status.idle":"2023-11-24T18:58:51.133363Z","shell.execute_reply.started":"2023-11-24T18:58:48.409003Z","shell.execute_reply":"2023-11-24T18:58:51.132421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_melted = df_de_train.melt(id_vars=['cell_type', 'sm_name'], value_vars=df_de_train.iloc[:,5:].columns, var_name='gene', value_name='value')\ntrain_df_melted","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:58:51.134884Z","iopub.execute_input":"2023-11-24T18:58:51.135179Z","iopub.status.idle":"2023-11-24T18:58:53.450617Z","shell.execute_reply.started":"2023-11-24T18:58:51.135154Z","shell.execute_reply":"2023-11-24T18:58:53.449488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_id_map = pd.read_csv(fn, index_col=0)\ncols_to_add = df_de_train.iloc[:,5:].columns\ndf_zeros = pd.DataFrame(0.0, columns=cols_to_add, index=df_id_map.index)\ntest_df = pd.concat([df_id_map, df_zeros], axis=1)\ntest_df","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:58:53.452394Z","iopub.execute_input":"2023-11-24T18:58:53.452695Z","iopub.status.idle":"2023-11-24T18:58:53.559366Z","shell.execute_reply.started":"2023-11-24T18:58:53.452668Z","shell.execute_reply":"2023-11-24T18:58:53.558413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_melted = test_df.melt(id_vars=['cell_type', 'sm_name'], value_vars=test_df.iloc[:,2:].columns, var_name='gene', value_name='value')\ntest_df_melted","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:58:54.251288Z","iopub.execute_input":"2023-11-24T18:58:54.251711Z","iopub.status.idle":"2023-11-24T18:58:56.113127Z","shell.execute_reply.started":"2023-11-24T18:58:54.251679Z","shell.execute_reply":"2023-11-24T18:58:56.112274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define Functions","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\n# from sklearn.decomposition import TruncatedSVD\nfrom sklearn.preprocessing import StandardScaler\n\ndef get_denoised_data(data, data_melted, n_comp=35):\n    Y = data.iloc[:,5:]\n\n    # Standardize the data\n    scaler = StandardScaler()\n    Y_std = scaler.fit_transform(Y)\n\n    reducer = PCA(n_components=n_comp, random_state=seed_value)\n\n    Yr_std = reducer.fit_transform(Y_std)\n    Yr_inv_trans_sdt = reducer.inverse_transform(Yr_std)\n    Yr_inv_trans_denorm = scaler.inverse_transform(Yr_inv_trans_sdt)\n    df_red_inv_trans = pd.DataFrame(Yr_inv_trans_denorm, columns = data.columns[5:])\n    df_de_train_denoised = pd.concat([data.iloc[:, :2], df_red_inv_trans], axis=1)\n    train_denoised_melted = df_de_train_denoised.melt(id_vars=['cell_type', 'sm_name'], value_vars=df_de_train_denoised.iloc[:,2:].columns, var_name='gene', value_name='value')\n    train_denoised_melted['is_valid'] = False\n    valid_indices = data_melted.sample(frac=0.2, random_state=seed_value).index.sort_values().tolist()\n    train_denoised_melted.loc[valid_indices, 'value'] = data_melted.loc[valid_indices, 'value']\n    train_denoised_melted.loc[valid_indices, 'is_valid'] = True\n    return train_denoised_melted","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:59:09.961599Z","iopub.execute_input":"2023-11-24T18:59:09.962383Z","iopub.status.idle":"2023-11-24T18:59:10.101614Z","shell.execute_reply.started":"2023-11-24T18:59:09.962352Z","shell.execute_reply":"2023-11-24T18:59:10.100829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def splitter(df):\n    train = df.index[~df['is_valid']].tolist()\n    valid = df.index[df['is_valid']].tolist()\n    return L(train), L(valid)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:59:10.822706Z","iopub.execute_input":"2023-11-24T18:59:10.823064Z","iopub.status.idle":"2023-11-24T18:59:10.828127Z","shell.execute_reply.started":"2023-11-24T18:59:10.823036Z","shell.execute_reply":"2023-11-24T18:59:10.827003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dls(data, data_melted, n_comp=None, bs=4096):\n    if n_comp:\n        train_denoised_melted = get_denoised_data(data, data_melted, n_comp=n_comp)\n        cats = train_denoised_melted.iloc[:, :3].columns.to_list()\n        splits = splitter(train_denoised_melted)\n        to = TabularPandas(train_denoised_melted, procs=[Categorify], cat_names=cats, y_names='value', splits=splits)\n        dls = to.dataloaders(bs)\n    else:\n        data_melted['is_valid'] = False\n        valid_indices = data_melted.sample(frac=0.2, random_state=seed_value).index.sort_values().tolist()\n        data_melted.loc[valid_indices, 'is_valid'] = True\n        cats = data_melted.iloc[:, :3].columns.to_list()\n        splits = splitter(data_melted)\n        to = TabularPandas(data_melted, procs=[Categorify], cat_names=cats, y_names='value', splits=splits)\n        dls = to.dataloaders(bs)\n    return dls","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:59:11.875105Z","iopub.execute_input":"2023-11-24T18:59:11.875469Z","iopub.status.idle":"2023-11-24T18:59:11.883396Z","shell.execute_reply.started":"2023-11-24T18:59:11.875439Z","shell.execute_reply":"2023-11-24T18:59:11.882367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rmse(preds, targs):\n    return ((targs-preds)**2).mean().sqrt().item()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:59:13.040903Z","iopub.execute_input":"2023-11-24T18:59:13.041255Z","iopub.status.idle":"2023-11-24T18:59:13.047268Z","shell.execute_reply.started":"2023-11-24T18:59:13.041227Z","shell.execute_reply":"2023-11-24T18:59:13.046474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train without PCA","metadata":{}},{"cell_type":"code","source":"%%time\ndls = get_dls(df_de_train, train_df_melted, n_comp=None, bs=4096)\ndls.valid.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:59:13.845447Z","iopub.execute_input":"2023-11-24T18:59:13.846210Z","iopub.status.idle":"2023-11-24T18:59:23.369261Z","shell.execute_reply.started":"2023-11-24T18:59:13.846164Z","shell.execute_reply":"2023-11-24T18:59:23.368195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.train.show()","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:59:23.370909Z","iopub.execute_input":"2023-11-24T18:59:23.371190Z","iopub.status.idle":"2023-11-24T18:59:23.472272Z","shell.execute_reply.started":"2023-11-24T18:59:23.371167Z","shell.execute_reply":"2023-11-24T18:59:23.471277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"emb_szs = {'cell_type': 5, 'sm_name': 26, 'gene': 1000}\nget_emb_sz(dls.train_ds)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:59:47.638599Z","iopub.execute_input":"2023-11-24T18:59:47.639413Z","iopub.status.idle":"2023-11-24T18:59:47.645903Z","shell.execute_reply.started":"2023-11-24T18:59:47.639382Z","shell.execute_reply":"2023-11-24T18:59:47.644973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = dls.train.y\ny_min = np.ceil(y.min()).astype(int)\ny_max = np.ceil(y.max()).astype(int)\ny_min, y_max","metadata":{"execution":{"iopub.status.busy":"2023-11-24T18:59:57.814207Z","iopub.execute_input":"2023-11-24T18:59:57.815074Z","iopub.status.idle":"2023-11-24T18:59:57.849706Z","shell.execute_reply.started":"2023-11-24T18:59:57.815042Z","shell.execute_reply":"2023-11-24T18:59:57.848751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = tabular_learner(dls, y_range=(y_min, y_max), emb_szs=emb_szs, layers=[1000, 500, 250], n_out=1, loss_func=F.mse_loss)\nlearn.fit_one_cycle(20, 3e-4)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:00:05.949622Z","iopub.execute_input":"2023-11-24T19:00:05.950307Z","iopub.status.idle":"2023-11-24T19:24:03.935507Z","shell.execute_reply.started":"2023-11-24T19:00:05.950276Z","shell.execute_reply":"2023-11-24T19:24:03.934436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds, targs = learn.get_preds()\nrmse(preds, targs)","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:24:03.937410Z","iopub.execute_input":"2023-11-24T19:24:03.938038Z","iopub.status.idle":"2023-11-24T19:24:10.884486Z","shell.execute_reply.started":"2023-11-24T19:24:03.938003Z","shell.execute_reply":"2023-11-24T19:24:10.883551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get Cell Embeddings","metadata":{}},{"cell_type":"code","source":"model = learn.model\nmodel.eval()\nmodel.to('cpu')","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:24:41.124286Z","iopub.execute_input":"2023-11-24T19:24:41.125120Z","iopub.status.idle":"2023-11-24T19:24:41.257429Z","shell.execute_reply.started":"2023-11-24T19:24:41.125084Z","shell.execute_reply":"2023-11-24T19:24:41.256479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_embs = learn.model.embeds[0].weight.detach().numpy()\ncell_embs","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:24:46.224267Z","iopub.execute_input":"2023-11-24T19:24:46.224618Z","iopub.status.idle":"2023-11-24T19:24:46.232466Z","shell.execute_reply.started":"2023-11-24T19:24:46.224564Z","shell.execute_reply":"2023-11-24T19:24:46.231487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.train.dataset.cat_names","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:24:48.720771Z","iopub.execute_input":"2023-11-24T19:24:48.721917Z","iopub.status.idle":"2023-11-24T19:24:48.728895Z","shell.execute_reply.started":"2023-11-24T19:24:48.721877Z","shell.execute_reply":"2023-11-24T19:24:48.727725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.train.dataset.classes['cell_type']","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:24:51.980854Z","iopub.execute_input":"2023-11-24T19:24:51.981674Z","iopub.status.idle":"2023-11-24T19:24:51.987448Z","shell.execute_reply.started":"2023-11-24T19:24:51.981642Z","shell.execute_reply":"2023-11-24T19:24:51.986467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab = dls.train.dataset.classes['cell_type']\ncell_emb_dict = {val: cell_embs[i] for i, val in enumerate(vocab)}\ncell_emb_df = pd.DataFrame(cell_emb_dict)\ncell_emb_df","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:01.267775Z","iopub.execute_input":"2023-11-24T19:25:01.268132Z","iopub.status.idle":"2023-11-24T19:25:01.284343Z","shell.execute_reply.started":"2023-11-24T19:25:01.268103Z","shell.execute_reply":"2023-11-24T19:25:01.283441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_emb_df = cell_emb_df.drop(columns=['#na#'])\ncell_emb_df","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:03.817346Z","iopub.execute_input":"2023-11-24T19:25:03.818089Z","iopub.status.idle":"2023-11-24T19:25:03.830530Z","shell.execute_reply.started":"2023-11-24T19:25:03.818053Z","shell.execute_reply":"2023-11-24T19:25:03.829526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_emb_df_melted = cell_emb_df.reset_index().melt(id_vars='index')\ncell_emb_df_melted","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:11.679254Z","iopub.execute_input":"2023-11-24T19:25:11.680037Z","iopub.status.idle":"2023-11-24T19:25:11.695139Z","shell.execute_reply.started":"2023-11-24T19:25:11.680001Z","shell.execute_reply":"2023-11-24T19:25:11.694243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_emb_df_pivoted = cell_emb_df_melted.pivot(index='variable', columns='index', values='value')\ncell_emb_df_pivoted.columns = [f'cemb{i}' for i in cell_emb_df_pivoted.columns]\ncell_emb_df_pivoted = cell_emb_df_pivoted.reset_index()\ncell_emb_df_pivoted = cell_emb_df_pivoted.rename(columns={'variable':'cell_type'})","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:20.102283Z","iopub.execute_input":"2023-11-24T19:25:20.102977Z","iopub.status.idle":"2023-11-24T19:25:20.116403Z","shell.execute_reply.started":"2023-11-24T19:25:20.102944Z","shell.execute_reply":"2023-11-24T19:25:20.115473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_emb_df_pivoted","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:26.170762Z","iopub.execute_input":"2023-11-24T19:25:26.171095Z","iopub.status.idle":"2023-11-24T19:25:26.184149Z","shell.execute_reply.started":"2023-11-24T19:25:26.171071Z","shell.execute_reply":"2023-11-24T19:25:26.183245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_emb_df_pivoted.to_csv('cell_embeddings_no_pca.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:31.215164Z","iopub.execute_input":"2023-11-24T19:25:31.215796Z","iopub.status.idle":"2023-11-24T19:25:31.222691Z","shell.execute_reply.started":"2023-11-24T19:25:31.215759Z","shell.execute_reply":"2023-11-24T19:25:31.221897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get Compund Embeddings","metadata":{}},{"cell_type":"code","source":"mol_embs = learn.model.embeds[1].weight.detach().numpy()\nvocab = dls.train.dataset.classes['sm_name']\nmol_emb_dict = {val: mol_embs[i] for i, val in enumerate(vocab)}\nmol_emb_df = pd.DataFrame(mol_emb_dict)\nmol_emb_df","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:34.618899Z","iopub.execute_input":"2023-11-24T19:25:34.619258Z","iopub.status.idle":"2023-11-24T19:25:34.673916Z","shell.execute_reply.started":"2023-11-24T19:25:34.619228Z","shell.execute_reply":"2023-11-24T19:25:34.672931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mol_emb_df = mol_emb_df.drop(columns=['#na#'])\nmol_emb_df_melted = mol_emb_df.reset_index().melt(id_vars='index')\nmol_emb_df_pivoted = mol_emb_df_melted.pivot(index='variable', columns='index', values='value')\nmol_emb_df_pivoted.columns = [f'memb{i}' for i in mol_emb_df_pivoted.columns]\nmol_emb_df_pivoted = mol_emb_df_pivoted.reset_index()\nmol_emb_df_pivoted = mol_emb_df_pivoted.rename(columns={'variable':'sm_name'})\nmol_emb_df_pivoted","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:34.894899Z","iopub.execute_input":"2023-11-24T19:25:34.895633Z","iopub.status.idle":"2023-11-24T19:25:34.943889Z","shell.execute_reply.started":"2023-11-24T19:25:34.895591Z","shell.execute_reply":"2023-11-24T19:25:34.942855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mol_emb_df_pivoted.to_csv('molecular_embeddings_no_pca.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:35.144213Z","iopub.execute_input":"2023-11-24T19:25:35.145049Z","iopub.status.idle":"2023-11-24T19:25:35.155378Z","shell.execute_reply.started":"2023-11-24T19:25:35.145015Z","shell.execute_reply":"2023-11-24T19:25:35.154643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get Gene Embeddings","metadata":{}},{"cell_type":"code","source":"gene_embs = learn.model.embeds[2].weight.detach().numpy()\nvocab = dls.train.dataset.classes['gene']\ngene_emb_dict = {val: gene_embs[i] for i, val in enumerate(vocab)}\ngene_emb_df = pd.DataFrame(gene_emb_dict)\ngene_emb_df","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:35.739963Z","iopub.execute_input":"2023-11-24T19:25:35.740794Z","iopub.status.idle":"2023-11-24T19:25:36.039301Z","shell.execute_reply.started":"2023-11-24T19:25:35.740761Z","shell.execute_reply":"2023-11-24T19:25:36.038317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gene_emb_df = gene_emb_df.drop(columns=['#na#'])\ngene_emb_df_melted = gene_emb_df.reset_index().melt(id_vars='index')\ngene_emb_df_pivoted = gene_emb_df_melted.pivot(index='variable', columns='index', values='value')\ngene_emb_df_pivoted.columns = [f'gemb{i}' for i in gene_emb_df_pivoted.columns]\ngene_emb_df_pivoted = gene_emb_df_pivoted.reset_index()\ngene_emb_df_pivoted = gene_emb_df_pivoted.rename(columns={'variable':'gene'})\ngene_emb_df_pivoted","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:36.040805Z","iopub.execute_input":"2023-11-24T19:25:36.041071Z","iopub.status.idle":"2023-11-24T19:25:46.031245Z","shell.execute_reply.started":"2023-11-24T19:25:36.041047Z","shell.execute_reply":"2023-11-24T19:25:46.030302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gene_emb_df_pivoted.to_csv('gene_embeddings_no_pca.csv')","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:25:46.032954Z","iopub.execute_input":"2023-11-24T19:25:46.033221Z","iopub.status.idle":"2023-11-24T19:26:11.522882Z","shell.execute_reply.started":"2023-11-24T19:25:46.033197Z","shell.execute_reply":"2023-11-24T19:26:11.522069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gene_emb_df_pivoted.to_parquet('gene_embeddings_no_pca.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-11-24T19:26:11.523962Z","iopub.execute_input":"2023-11-24T19:26:11.524220Z","iopub.status.idle":"2023-11-24T19:26:12.830545Z","shell.execute_reply.started":"2023-11-24T19:26:11.524198Z","shell.execute_reply":"2023-11-24T19:26:12.829742Z"},"trusted":true},"execution_count":null,"outputs":[]}]}