{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport pickle as pkl\nfrom collections import defaultdict\nimport matplotlib.pyplot as plt\nimport catboost as cb\nimport lightgbm as lgb\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-20T08:16:09.436720Z","iopub.execute_input":"2023-01-20T08:16:09.437581Z","iopub.status.idle":"2023-01-20T08:16:11.800252Z","shell.execute_reply.started":"2023-01-20T08:16:09.437478Z","shell.execute_reply":"2023-01-20T08:16:11.798341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\nrename=pd.read_csv('/kaggle/input/research-project-01-around-multimodal-singlecell/CD_info.csv')\nrna.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:16:11.802851Z","iopub.execute_input":"2023-01-20T08:16:11.803285Z","iopub.status.idle":"2023-01-20T08:16:54.305115Z","shell.execute_reply.started":"2023-01-20T08:16:11.803249Z","shell.execute_reply":"2023-01-20T08:16:54.304251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna.columns=rna.columns.str.split('_').str[0]\nrna=rna.T\nrename_copy=rename.rename(columns={'ENSG ID':'gene_id'})\nrename_copy=rename_copy.merge(rna, on='gene_id', how='inner')\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:16:54.306223Z","iopub.execute_input":"2023-01-20T08:16:54.307014Z","iopub.status.idle":"2023-01-20T08:16:57.903962Z","shell.execute_reply.started":"2023-01-20T08:16:54.306987Z","shell.execute_reply":"2023-01-20T08:16:57.902835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rename_copy.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:16:57.907727Z","iopub.execute_input":"2023-01-20T08:16:57.909099Z","iopub.status.idle":"2023-01-20T08:16:57.917701Z","shell.execute_reply.started":"2023-01-20T08:16:57.909056Z","shell.execute_reply":"2023-01-20T08:16:57.915957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rename_copy=rename_copy.drop(['Unnamed: 0','cd_names', 'names'], axis=1)\ntarget=rename_copy.T\ntarget.columns = target.iloc[0]\ntarget=target[1:]\ntarget.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:16:57.919386Z","iopub.execute_input":"2023-01-20T08:16:57.919872Z","iopub.status.idle":"2023-01-20T08:16:59.296107Z","shell.execute_reply.started":"2023-01-20T08:16:57.919816Z","shell.execute_reply":"2023-01-20T08:16:59.294070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def intersaction_coeff(list1, list2):\n    return len(set(list1).intersection(set(list2)))","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:16:59.298566Z","iopub.execute_input":"2023-01-20T08:16:59.299101Z","iopub.status.idle":"2023-01-20T08:16:59.311918Z","shell.execute_reply.started":"2023-01-20T08:16:59.299069Z","shell.execute_reply":"2023-01-20T08:16:59.308750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cos_sim(a, b):\n    cos_sim = np.dot(a, b)/(np.linalg.norm(a)*np.linalg.norm(b))\n    return cos_sim","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:16:59.315431Z","iopub.execute_input":"2023-01-20T08:16:59.316199Z","iopub.status.idle":"2023-01-20T08:16:59.327889Z","shell.execute_reply.started":"2023-01-20T08:16:59.316132Z","shell.execute_reply":"2023-01-20T08:16:59.325267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna=rna.T\nrna.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:16:59.330367Z","iopub.execute_input":"2023-01-20T08:16:59.331055Z","iopub.status.idle":"2023-01-20T08:16:59.423316Z","shell.execute_reply.started":"2023-01-20T08:16:59.330954Z","shell.execute_reply":"2023-01-20T08:16:59.419791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport catboost as cb\nimport gc\n\nX=rna\ny=target\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:16:59.428618Z","iopub.execute_input":"2023-01-20T08:16:59.429747Z","iopub.status.idle":"2023-01-20T08:17:10.729157Z","shell.execute_reply.started":"2023-01-20T08:16:59.429627Z","shell.execute_reply":"2023-01-20T08:17:10.726087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score, mean_squared_error\ndf_scores=pd.DataFrame()\nfor i in target.columns[90:100]:\n    print(i)\n    model = cb.CatBoostRegressor()\n    model.fit(X_train.loc[:,~X_train.columns.isin([i])], y_train[i])\n    model_dir_path = f\"/kaggle/working/models/{i}\"\n    os.makedirs(model_dir_path, exist_ok=True)\n\n    model_path = f\"/kaggle/working/models/{i}/model.cbm\"\n    model.save_model(model_path)\n    y_true=y_test[i]\n    y_pred = model.predict(X_test.loc[:,~X_test.columns.isin([i])])\n\n    s = r2_score(y_true,y_pred)\n    df_scores.loc['r2', i ]  = s\n    s = mean_squared_error(y_true,y_pred, squared=False) # squared=False -> RMSE, not MSE \n    df_scores.loc['RMSE', i ]  = s\n    print(f\"{i}, corr:\", df_scores)\n\n    del y_true\n    del y_pred\n    del model\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:17:10.734108Z","iopub.execute_input":"2023-01-20T08:17:10.734578Z","iopub.status.idle":"2023-01-20T16:51:34.515845Z","shell.execute_reply.started":"2023-01-20T08:17:10.734544Z","shell.execute_reply":"2023-01-20T16:51:34.513199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores.to_csv('df_scores_90-100.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-20T16:51:34.518779Z","iopub.execute_input":"2023-01-20T16:51:34.519361Z","iopub.status.idle":"2023-01-20T16:51:34.539536Z","shell.execute_reply.started":"2023-01-20T16:51:34.519304Z","shell.execute_reply":"2023-01-20T16:51:34.536990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}