{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"papermill":{"default_parameters":{},"duration":1767.896818,"end_time":"2023-10-03T19:16:48.560142","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2023-10-03T18:47:20.663324","version":"2.4.0"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":7123483,"sourceType":"datasetVersion","datasetId":4109071}],"dockerImageVersionId":30588,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import time\nt0start = time.time()\nfrom fastai.collab import *\nfrom fastai.tabular.all import *\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":6.509942,"end_time":"2023-10-03T18:47:30.760703","exception":false,"start_time":"2023-10-03T18:47:24.250761","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:55:35.199062Z","iopub.execute_input":"2023-12-04T19:55:35.199361Z","iopub.status.idle":"2023-12-04T19:55:41.236975Z","shell.execute_reply.started":"2023-12-04T19:55:35.199334Z","shell.execute_reply":"2023-12-04T19:55:41.235922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install -q xgboost","metadata":{"execution":{"iopub.status.busy":"2023-12-04T19:55:41.238844Z","iopub.execute_input":"2023-12-04T19:55:41.239285Z","iopub.status.idle":"2023-12-04T19:55:54.039057Z","shell.execute_reply.started":"2023-12-04T19:55:41.239257Z","shell.execute_reply":"2023-12-04T19:55:54.037762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_seed = 177","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:55:54.040549Z","iopub.execute_input":"2023-12-04T19:55:54.040874Z","iopub.status.idle":"2023-12-04T19:55:54.046184Z","shell.execute_reply.started":"2023-12-04T19:55:54.040845Z","shell.execute_reply":"2023-12-04T19:55:54.045247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading and Melting Train and Test Data","metadata":{"tags":[]}},{"cell_type":"markdown","source":"Here I read the training and test data and melt it to yield a ```DataFrame``` with three categorical features (```cell_type```, ```sm_name```, and ```gene```) and one target (```value```).","metadata":{}},{"cell_type":"code","source":"%%time\nfn = '/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet'\ndf_de_train = pd.read_parquet(fn)# , index_col = 0)\ntrain_df = df_de_train.melt(id_vars=['cell_type', 'sm_name'], value_vars=df_de_train.iloc[:,5:].columns, var_name='gene', value_name='value')","metadata":{"papermill":{"duration":163.861162,"end_time":"2023-10-03T18:50:18.214746","exception":false,"start_time":"2023-10-03T18:47:34.353584","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:55:54.048834Z","iopub.execute_input":"2023-12-04T19:55:54.049484Z","iopub.status.idle":"2023-12-04T19:55:58.839634Z","shell.execute_reply.started":"2023-12-04T19:55:54.049452Z","shell.execute_reply":"2023-12-04T19:55:58.838683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fn = '/kaggle/input/open-problems-single-cell-perturbations/id_map.csv'\ndf_id_map = pd.read_csv(fn)\nfn = '/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv'\ndf = pd.read_csv(fn, index_col = 0)\n\ncols_to_add = df_de_train.iloc[:,5:].columns\ncols_to_add\n\ndf_zeros = pd.DataFrame(0.0, columns=cols_to_add, index=df_id_map.index)\ndf_zeros\n\ndf_id_map_preds = pd.concat([df_id_map, df_zeros], axis=1)\ntest_df = df_id_map_preds.melt(id_vars=['cell_type', 'sm_name'], value_vars=df_id_map_preds.iloc[:,3:].columns, var_name='gene', value_name='value')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:55:58.840951Z","iopub.execute_input":"2023-12-04T19:55:58.841335Z","iopub.status.idle":"2023-12-04T19:56:03.537942Z","shell.execute_reply.started":"2023-12-04T19:55:58.841300Z","shell.execute_reply":"2023-12-04T19:56:03.537159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Embeddings","metadata":{}},{"cell_type":"markdown","source":"The function ```reduce_emb_dim``` is used to reduce the dimensionality of molecular (```dim = 26```) and gene embeddings (```dim = 1000```) down to 10 components.","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler\n\ndef reduce_emb_dim(data, n_comp=35, random_state=42):\n    embname = data.columns[1][:-1]\n    Y = data.iloc[:,1:]\n    scaler = StandardScaler()\n    Y_std = scaler.fit_transform(Y)\n    reducer = PCA(n_components=n_comp, random_state=random_state)\n\n    Yr = reducer.fit_transform(Y_std)\n    column_names = [f'{embname}pca{n_comp}_{i}' for i in range(n_comp)]\n    reduced_data = pd.DataFrame(Yr, columns = column_names)\n    reduced_data = pd.concat([data.iloc[:, 0], reduced_data], axis=1)\n    return reduced_data","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:56:03.539347Z","iopub.execute_input":"2023-12-04T19:56:03.540086Z","iopub.status.idle":"2023-12-04T19:56:03.643670Z","shell.execute_reply.started":"2023-12-04T19:56:03.540047Z","shell.execute_reply":"2023-12-04T19:56:03.642680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_embs = pd.read_csv('/kaggle/input/op2-single-cell-perturbations-tabmodnn-embeddings/cell_embeddings_no_pca.csv', index_col = 0)\nmol_embs = pd.read_csv('/kaggle/input/op2-single-cell-perturbations-tabmodnn-embeddings/molecular_embeddings_no_pca.csv', index_col = 0)\ngene_embs = pd.read_parquet('/kaggle/input/op2-single-cell-perturbations-tabmodnn-embeddings/gene_embeddings_no_pca.parquet')\ncell_embs.shape, mol_embs.shape, gene_embs.shape","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:56:03.644812Z","iopub.execute_input":"2023-12-04T19:56:03.645095Z","iopub.status.idle":"2023-12-04T19:56:04.748381Z","shell.execute_reply.started":"2023-12-04T19:56:03.645070Z","shell.execute_reply":"2023-12-04T19:56:04.747437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gene_embs = reduce_emb_dim(gene_embs, n_comp=10, random_state=random_seed)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:56:04.749432Z","iopub.execute_input":"2023-12-04T19:56:04.749707Z","iopub.status.idle":"2023-12-04T19:56:06.057793Z","shell.execute_reply.started":"2023-12-04T19:56:04.749682Z","shell.execute_reply":"2023-12-04T19:56:06.056183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mol_embs = reduce_emb_dim(mol_embs, n_comp=10, random_state=random_seed)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:56:06.060044Z","iopub.execute_input":"2023-12-04T19:56:06.060720Z","iopub.status.idle":"2023-12-04T19:56:06.178875Z","shell.execute_reply.started":"2023-12-04T19:56:06.060666Z","shell.execute_reply":"2023-12-04T19:56:06.178015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define Functions","metadata":{"tags":[]}},{"cell_type":"markdown","source":"The function ```get_train_valid_data``` takes the train and test data as inputs and returns a ```TabularPandas``` object using the ```valid``` scheme and ```split_func``` provided.","metadata":{}},{"cell_type":"code","source":"def splitter(df):\n    train = df.index[~df['is_valid']].tolist()\n    valid = df.index[df['is_valid']].tolist()\n    return L(train), L(valid)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T19:56:06.182408Z","iopub.execute_input":"2023-12-04T19:56:06.182727Z","iopub.status.idle":"2023-12-04T19:56:06.188545Z","shell.execute_reply.started":"2023-12-04T19:56:06.182700Z","shell.execute_reply":"2023-12-04T19:56:06.187493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_valid_data(train_data, test_data, valid=None, random_state=None, split_func=splitter):\n    train_data['is_valid'] = False\n    cont, cat = cont_cat_split(train_data, 1, dep_var='value')\n    if valid in ['NK cells', 'T cells CD4+', 'T cells CD8+', 'T regulatory cells']:\n        test_compounds = test_data['sm_name'].unique().tolist()\n        valid_indices = train_data.loc[(train_data['cell_type']==valid) & train_data['sm_name'].isin(test_compounds)].index.sort_values().tolist()\n        train_data.loc[valid_indices, 'is_valid'] = True\n        splits = splitter(train_data)\n        cont, cat = cont_cat_split(train_data, 1, dep_var='value')\n        cat.remove('is_valid')\n        to = TabularPandas(train_data, procs=[Categorify, FillMissing], cont_names=cont, cat_names=cat, y_names='value', splits=splits)\n    elif type(valid)==float:\n        valid_indices = train_data.sample(frac=valid, random_state=random_state).index.sort_values().tolist()\n        train_data.loc[valid_indices, 'is_valid'] = True\n        splits = splitter(train_data)\n        cat.remove('is_valid')\n        to = TabularPandas(train_data, procs=[Categorify, FillMissing], cont_names=cont, cat_names=cat, y_names='value', splits=splits)\n    else:\n        cat.remove('is_valid')\n        to = TabularPandas(train_data, procs=[Categorify, FillMissing], cont_names=cont, cat_names=cat, y_names='value', splits=None)\n    return to","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:56:06.189956Z","iopub.execute_input":"2023-12-04T19:56:06.190413Z","iopub.status.idle":"2023-12-04T19:56:06.201962Z","shell.execute_reply.started":"2023-12-04T19:56:06.190374Z","shell.execute_reply":"2023-12-04T19:56:06.201113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rmse(preds, targs): return round(math.sqrt(((targs-preds)**2).mean()), 6)\ndef m_rmse(m, xs, y): return rmse(m.predict(xs), y)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T19:56:06.203860Z","iopub.execute_input":"2023-12-04T19:56:06.204151Z","iopub.status.idle":"2023-12-04T19:56:06.215234Z","shell.execute_reply.started":"2023-12-04T19:56:06.204126Z","shell.execute_reply":"2023-12-04T19:56:06.214375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Add Embeddings","metadata":{}},{"cell_type":"markdown","source":"The following lines merge the embeddings with the train and test data.","metadata":{}},{"cell_type":"code","source":"train_df_c = pd.merge(train_df, cell_embs, on='cell_type', how='left')\ntrain_df_cm = pd.merge(train_df_c, mol_embs, on='sm_name', how='left')\ntrain_df_cmg = pd.merge(train_df_cm, gene_embs, on='gene', how='left')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:56:06.216316Z","iopub.execute_input":"2023-12-04T19:56:06.216717Z","iopub.status.idle":"2023-12-04T19:56:16.512330Z","shell.execute_reply.started":"2023-12-04T19:56:06.216690Z","shell.execute_reply":"2023-12-04T19:56:16.511338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_c = pd.merge(test_df, cell_embs, on='cell_type', how='left')\ntest_df_cm = pd.merge(test_df_c, mol_embs, on='sm_name', how='left')\ntest_df_cmg = pd.merge(test_df_cm, gene_embs, on='gene', how='left')","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:56:16.513535Z","iopub.execute_input":"2023-12-04T19:56:16.513827Z","iopub.status.idle":"2023-12-04T19:56:20.820325Z","shell.execute_reply.started":"2023-12-04T19:56:16.513801Z","shell.execute_reply":"2023-12-04T19:56:20.819417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Delete Unused Data","metadata":{}},{"cell_type":"markdown","source":"And collect garbage to save some memory.","metadata":{}},{"cell_type":"code","source":"import gc\ndel train_df_c\ndel train_df_cm\ndel test_df_c\ndel test_df_cm\ndel cell_embs\ndel mol_embs\ndel gene_embs\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T19:56:20.821502Z","iopub.execute_input":"2023-12-04T19:56:20.821797Z","iopub.status.idle":"2023-12-04T19:56:21.269935Z","shell.execute_reply.started":"2023-12-04T19:56:20.821770Z","shell.execute_reply":"2023-12-04T19:56:21.268916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Hierarchie","metadata":{}},{"cell_type":"markdown","source":"Convert data and check features for redundancy.","metadata":{}},{"cell_type":"code","source":"%%time\nto = get_train_valid_data(train_df_cmg, test_df_cmg, valid=None, random_state=random_seed)\nxs, y = to.train.xs, to.train.y","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:56:21.271059Z","iopub.execute_input":"2023-12-04T19:56:21.271369Z","iopub.status.idle":"2023-12-04T19:56:34.330778Z","shell.execute_reply.started":"2023-12-04T19:56:21.271344Z","shell.execute_reply":"2023-12-04T19:56:34.329824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfrom scipy.cluster import hierarchy\nfrom scipy.spatial.distance import squareform\ncorrelation_matrix = xs.corr()\ndistances = 1 - np.abs(correlation_matrix)\nnp.fill_diagonal(distances.values, 0)\ncondensed_distances = squareform(distances)\nlinkage_matrix = hierarchy.linkage(condensed_distances, method='complete')\nplt.figure(figsize=(10,10))\nhierarchy.dendrogram(linkage_matrix, labels=xs.columns.tolist(), orientation='left')\nplt.show()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:56:34.331979Z","iopub.execute_input":"2023-12-04T19:56:34.332284Z","iopub.status.idle":"2023-12-04T19:56:56.506830Z","shell.execute_reply.started":"2023-12-04T19:56:34.332257Z","shell.execute_reply":"2023-12-04T19:56:56.505836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train XGBoosted Forest","metadata":{}},{"cell_type":"code","source":"%%time\nimport xgboost as xgb\nm = xgb.XGBRegressor(device='cuda', n_estimators=185, learning_rate=0.01, max_depth=20, min_child_weight=1, gamma=0, subsample=1, colsample_bytree=0.55, random_state=random_seed) # reg_lambda=10, alpha=10\nm.fit(xs, y)\nm_rmse(m, xs, y)","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T19:56:56.507983Z","iopub.execute_input":"2023-12-04T19:56:56.508281Z","iopub.status.idle":"2023-12-04T20:11:43.923375Z","shell.execute_reply.started":"2023-12-04T19:56:56.508255Z","shell.execute_reply":"2023-12-04T20:11:43.922446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference and Submission","metadata":{}},{"cell_type":"code","source":"test_to = to.new(test_df_cmg)\ntest_to.process()\ntest_xs = test_to.xs\ntest_xs","metadata":{"papermill":{"duration":7.510266,"end_time":"2023-10-03T19:15:32.667807","exception":false,"start_time":"2023-10-03T19:15:25.157541","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-04T20:11:43.924822Z","iopub.execute_input":"2023-12-04T20:11:43.925461Z","iopub.status.idle":"2023-12-04T20:11:46.460454Z","shell.execute_reply.started":"2023-12-04T20:11:43.925426Z","shell.execute_reply":"2023-12-04T20:11:46.459479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = tensor(m.predict(test_xs))\npreds","metadata":{"papermill":{"duration":65.204612,"end_time":"2023-10-03T19:16:37.884386","exception":false,"start_time":"2023-10-03T19:15:32.679774","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-04T20:11:46.461607Z","iopub.execute_input":"2023-12-04T20:11:46.461881Z","iopub.status.idle":"2023-12-04T20:11:51.525573Z","shell.execute_reply.started":"2023-12-04T20:11:46.461858Z","shell.execute_reply":"2023-12-04T20:11:51.524495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.min(), preds.max()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2023-12-04T20:11:51.526679Z","iopub.execute_input":"2023-12-04T20:11:51.526934Z","iopub.status.idle":"2023-12-04T20:11:51.536026Z","shell.execute_reply.started":"2023-12-04T20:11:51.526912Z","shell.execute_reply":"2023-12-04T20:11:51.535138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the last step I reshape the predictions back into a 255 x 18211 tensor for submission:","metadata":{}},{"cell_type":"code","source":"to_submit = preds.view(18211, -1).t().numpy()\nsubmit = pd.DataFrame(to_submit, columns=df_de_train.iloc[:,5:].columns)\nsubmit.index.name = 'id'\nsubmit","metadata":{"papermill":{"duration":0.152745,"end_time":"2023-10-03T19:16:38.048033","exception":false,"start_time":"2023-10-03T19:16:37.895288","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-04T20:11:51.537380Z","iopub.execute_input":"2023-12-04T20:11:51.538284Z","iopub.status.idle":"2023-12-04T20:11:51.591494Z","shell.execute_reply.started":"2023-12-04T20:11:51.538255Z","shell.execute_reply":"2023-12-04T20:11:51.590597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('submission.csv')","metadata":{"papermill":{"duration":7.117791,"end_time":"2023-10-03T19:16:45.177690","exception":false,"start_time":"2023-10-03T19:16:38.059899","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-12-04T20:11:51.592897Z","iopub.execute_input":"2023-12-04T20:11:51.593816Z","iopub.status.idle":"2023-12-04T20:11:59.641180Z","shell.execute_reply.started":"2023-12-04T20:11:51.593778Z","shell.execute_reply":"2023-12-04T20:11:59.640415Z"},"trusted":true},"execution_count":null,"outputs":[]}]}