{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### The purpose of this notebook is just to get you quickly started and be familiar with the data. In the cite part, I used a RandomForestRegressor to predict the effect on protein expression level by CELL_ID which reveals information about donor, days, cell_type, batch effect and etc.. The effect of gene_id related to gene loci is simply assigned as the mean expression of the protein. I used the formula 0.2*RF_prediction + 0.8*mean_protein_expression. \n\n### For the multi part, we can use the similar formula. But due to the RAM and CPU limitation, I just used the formula 1*mean_protein_expression. For details on how to do average by gene_id, please check another kaggler's notebook: https://www.kaggle.com/code/shuntarotanaka/simple-submission-average-by-gene-id  Thank him very much! In his notebook, he also converted strings to int to save memory usage. ","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.metrics import log_loss\nfrom pathlib import Path\npath = Path('/kaggle/input/open-problems-multimodal/')","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:07:10.474819Z","iopub.execute_input":"2022-12-01T02:07:10.475512Z","iopub.status.idle":"2022-12-01T02:07:11.019922Z","shell.execute_reply.started":"2022-12-01T02:07:10.475446Z","shell.execute_reply":"2022-12-01T02:07:11.018500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########Cite ################\n########read cite files#########\nSTART = int(1e4)\n#print(START)\nSTOP = START+1000\nx_train_cite=pd.DataFrame(pd.read_hdf(path/\"train_cite_inputs.h5\", start=START, stop=STOP))\ny_train_cite=pd.DataFrame(pd.read_hdf(path/\"train_cite_targets.h5\", start=START, stop=STOP))\nx_test_cite=pd.DataFrame(pd.read_hdf(path/\"test_cite_inputs.h5\"))\n\n\n\n# print(x_train_cite.info)\n# print(y_train_cite.info)\n# print(x_test_cite.info)\n#print(y_train_cite_mean)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:07:11.022119Z","iopub.execute_input":"2022-12-01T02:07:11.022511Z","iopub.status.idle":"2022-12-01T02:07:50.910601Z","shell.execute_reply.started":"2022-12-01T02:07:11.022477Z","shell.execute_reply":"2022-12-01T02:07:50.908997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#########caculating mean based on cell ID or gene ID##########\nx_train_cite_mean=pd.DataFrame(x_train_cite.mean(axis=1))\ny_train_cite_mean=pd.DataFrame(y_train_cite.mean(axis=1))\nx_test_cite_mean=pd.DataFrame(x_test_cite.mean(axis=1))\ngene_cite_mean=pd.DataFrame(y_train_cite.mean(axis=0))\n\n# x_train_cite_mean[\"cell_id\"]=x_train_cite_mean.index\n# y_train_cite_mean[\"cell_id\"]=y_train_cite_mean.index\n# x_test_cite_mean[\"cell_id\"]=x_test_cite_mean.index\nprint(x_train_cite_mean.info)\nprint(y_train_cite_mean.info)\n# print(x_train_cite_mean.index)\n#print(y_train_cite_mean)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:07:50.912314Z","iopub.execute_input":"2022-12-01T02:07:50.912687Z","iopub.status.idle":"2022-12-01T02:07:53.679133Z","shell.execute_reply.started":"2022-12-01T02:07:50.912653Z","shell.execute_reply":"2022-12-01T02:07:53.678000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gene_cite_mean = gene_cite_mean.rename(columns={0: 'target'})\ngene_cite_mean[\"gene_id\"]=gene_cite_mean.index\nprint(gene_cite_mean)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:07:53.681106Z","iopub.execute_input":"2022-12-01T02:07:53.681802Z","iopub.status.idle":"2022-12-01T02:07:53.698692Z","shell.execute_reply.started":"2022-12-01T02:07:53.681758Z","shell.execute_reply":"2022-12-01T02:07:53.697230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#############RandomForest for mean by cell ID###########\n\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.ensemble import RandomForestRegressor\n#from sklearn.decomposition import PCA\nmodel_clf = RandomForestRegressor(max_depth=2, random_state=0) \nmodel_clf.fit(x_train_cite_mean, y_train_cite_mean)\ny_test_cite_mean=pd.DataFrame(model_clf.predict(x_test_cite_mean))\ny_test_cite_mean[\"cell_id\"]=x_test_cite.index\n\ny_test_cite_mean = y_test_cite_mean.rename(columns={0: 'target'})\nprint(y_test_cite_mean)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:07:53.701835Z","iopub.execute_input":"2022-12-01T02:07:53.703030Z","iopub.status.idle":"2022-12-01T02:07:54.075930Z","shell.execute_reply.started":"2022-12-01T02:07:53.702988Z","shell.execute_reply":"2022-12-01T02:07:54.074753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###################combine y_test_cite_mean and gene_cite_mean##########\n####cell_id + gene_id = 1/2*cell_id_value + 1/2*gene_id_value\n\n\nnew_target=[]\n\nfor e in y_test_cite_mean[\"target\"]:\n    for f in gene_cite_mean[\"target\"]:\n        new_target.append(0.2*e+0.8*f)\n        \n        \n        \n#         new_id.append(e+f)\n#         new_target.append(0.2*(y_test_cite_mean[\"cell_id\"]\n#                           .loc[y_test_cite_mean[\"cell_id\"]==e, \"target\"])\n#                           +0.8*(gene_cite_mean[\"gene_id\"]\n#                           .loc[gene_cite_mean[\"gene_id\"]==f, \"target\"]))\n\n\n\nprint(len(new_target))      \n\n","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:07:54.077649Z","iopub.execute_input":"2022-12-01T02:07:54.078849Z","iopub.status.idle":"2022-12-01T02:07:56.471390Z","shell.execute_reply.started":"2022-12-01T02:07:54.078808Z","shell.execute_reply":"2022-12-01T02:07:56.469934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###############muliome#############\n#############read multi files and calcuate means#########\nSTART = int(1e4)\nprint(START)\nSTOP = START+10000\n\n#x_train_multi=pd.read_hdf(path/\"train_multi_inputs.h5\", start=START, stop=STOP)\ny_train_multi=pd.read_hdf(path/\"train_multi_targets.h5\", start=START, stop=STOP)\n\n\n# y_train_multi_mean=pd.DataFrame(y_train_multi.mean(axis=1))\n# x_train_multi_mean=pd.DataFrame(x_train_multi.mean(axis=1))\n# x_test_multi_mean=pd.DataFrame(x_test_multi.mean(axis=1))\n\ngene_multi_mean=pd.DataFrame(y_train_multi.mean(axis=0))\n\ngene_multi_mean=gene_multi_mean.rename(columns={0: 'target'})\n\n\n# print(y_train_multi_mean.info)\n# print(x_train_multi_mean.info)\n# print(x_test_multi_mean.info)\nprint(gene_multi_mean)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:07:56.473906Z","iopub.execute_input":"2022-12-01T02:07:56.474720Z","iopub.status.idle":"2022-12-01T02:08:03.903410Z","shell.execute_reply.started":"2022-12-01T02:07:56.474670Z","shell.execute_reply":"2022-12-01T02:08:03.902224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ############RandomForest for mean based on cell ID#############\n# model_clf2 = RandomForestRegressor(max_depth=2, random_state=0) \n# model_clf2.fit(x_train_multi_mean, y_train_multi_mean)\n# y_test_multi_mean=pd.DataFrame(model_clf2.predict(x_test_multi_mean))\n# y_test_multi_mean[\"cell_id\"]=x_test_multi.index\n# y_test_multi_mean = y_test_multi_mean.rename(columns={0: 'target'})\n# print(y_test_multi_mean)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:08:03.904837Z","iopub.execute_input":"2022-12-01T02:08:03.905177Z","iopub.status.idle":"2022-12-01T02:08:03.910534Z","shell.execute_reply.started":"2022-12-01T02:08:03.905138Z","shell.execute_reply":"2022-12-01T02:08:03.909645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ###################combine y_test_cite_mean and gene_cite_mean##########\n# ####cell_id + gene_id = 1/2*cell_id_value + 1/2*gene_id_value\n\n\n# new_target2=[]\n\n# for e in y_test_multi_mean[\"target\"]:\n#     for f in gene_multi_mean[\"target\"]:\n#         new_target2.append(0.2*e+0.8*f)\n        \n        \n        \n# #         new_id.append(e+f)\n# #         new_target.append(0.2*(y_test_cite_mean[\"cell_id\"]\n# #                           .loc[y_test_cite_mean[\"cell_id\"]==e, \"target\"])\n# #                           +0.8*(gene_cite_mean[\"gene_id\"]\n# #                           .loc[gene_cite_mean[\"gene_id\"]==f, \"target\"]))\n\n\n\n# print(len(new_target2))  ","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:08:03.912142Z","iopub.execute_input":"2022-12-01T02:08:03.912460Z","iopub.status.idle":"2022-12-01T02:08:03.933151Z","shell.execute_reply.started":"2022-12-01T02:08:03.912432Z","shell.execute_reply":"2022-12-01T02:08:03.931398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_evaluation=pd.read_csv(path/\"evaluation_ids.csv\")\ndf_evaluation2=df_evaluation[6812820:]\n\nprint(len(df_evaluation))","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:08:03.935314Z","iopub.execute_input":"2022-12-01T02:08:03.937415Z","iopub.status.idle":"2022-12-01T02:09:11.618625Z","shell.execute_reply.started":"2022-12-01T02:08:03.937358Z","shell.execute_reply":"2022-12-01T02:09:11.617096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_evaluation2=df_evaluation2[[\"row_id\",\"gene_id\"]]\nprint(df_evaluation.head(5))","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:09:11.620343Z","iopub.execute_input":"2022-12-01T02:09:11.621761Z","iopub.status.idle":"2022-12-01T02:09:12.819750Z","shell.execute_reply.started":"2022-12-01T02:09:11.621717Z","shell.execute_reply":"2022-12-01T02:09:12.817774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission2=df_evaluation2.merge(gene_multi_mean, how=\"left\", on=\"gene_id\")\nprint(df_submission2)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:23:33.611937Z","iopub.execute_input":"2022-12-01T02:23:33.612484Z","iopub.status.idle":"2022-12-01T02:23:59.155369Z","shell.execute_reply.started":"2022-12-01T02:23:33.612448Z","shell.execute_reply":"2022-12-01T02:23:59.153680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"###############concact cite and multi###########\ndf_submission1=pd.DataFrame({\"target\":new_target})\ndf=pd.concat( [df_submission1[\"target\"], df_submission2[\"target\"]], ignore_index=True)\nprint(df.head(5))\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:20:42.461028Z","iopub.execute_input":"2022-12-01T02:20:42.461620Z","iopub.status.idle":"2022-12-01T02:20:44.390281Z","shell.execute_reply.started":"2022-12-01T02:20:42.461576Z","shell.execute_reply":"2022-12-01T02:20:44.388937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().values.any()","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:09:35.793605Z","iopub.execute_input":"2022-12-01T02:09:35.794006Z","iopub.status.idle":"2022-12-01T02:09:35.878203Z","shell.execute_reply.started":"2022-12-01T02:09:35.793976Z","shell.execute_reply":"2022-12-01T02:09:35.876645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.DataFrame(df)\ndf=df.reset_index()\nprint(df)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:20:50.261180Z","iopub.execute_input":"2022-12-01T02:20:50.261675Z","iopub.status.idle":"2022-12-01T02:20:51.221150Z","shell.execute_reply.started":"2022-12-01T02:20:50.261636Z","shell.execute_reply":"2022-12-01T02:20:51.219391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=df.rename(columns={\"index\":\"row_id\"})\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2022-12-01T02:22:35.201581Z","iopub.execute_input":"2022-12-01T02:22:35.202119Z","iopub.status.idle":"2022-12-01T02:22:35.914651Z","shell.execute_reply.started":"2022-12-01T02:22:35.202072Z","shell.execute_reply":"2022-12-01T02:22:35.912908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('submission.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]}]}