{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport pickle as pkl\nfrom collections import defaultdict\nimport matplotlib.pyplot as plt\nimport catboost as cb\nimport lightgbm as lgb\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-20T07:39:05.772212Z","iopub.execute_input":"2023-01-20T07:39:05.773829Z","iopub.status.idle":"2023-01-20T07:39:08.978316Z","shell.execute_reply.started":"2023-01-20T07:39:05.773603Z","shell.execute_reply":"2023-01-20T07:39:08.977142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\nrename=pd.read_csv('/kaggle/input/research-project-01-around-multimodal-singlecell/CD_info.csv')\nrna.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T07:39:08.981006Z","iopub.execute_input":"2023-01-20T07:39:08.981813Z","iopub.status.idle":"2023-01-20T07:40:15.000624Z","shell.execute_reply.started":"2023-01-20T07:39:08.981764Z","shell.execute_reply":"2023-01-20T07:40:14.999460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna.columns=rna.columns.str.split('_').str[0]\nrna=rna.T\nrename_copy=rename.rename(columns={'ENSG ID':'gene_id'})\nrename_copy=rename_copy.merge(rna, on='gene_id', how='inner')\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T07:40:15.002158Z","iopub.execute_input":"2023-01-20T07:40:15.002987Z","iopub.status.idle":"2023-01-20T07:40:27.108815Z","shell.execute_reply.started":"2023-01-20T07:40:15.002943Z","shell.execute_reply":"2023-01-20T07:40:27.107470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rename_copy.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-20T07:40:27.111773Z","iopub.execute_input":"2023-01-20T07:40:27.112227Z","iopub.status.idle":"2023-01-20T07:40:27.119544Z","shell.execute_reply.started":"2023-01-20T07:40:27.112190Z","shell.execute_reply":"2023-01-20T07:40:27.118460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rename_copy=rename_copy.drop(['Unnamed: 0','cd_names', 'names'], axis=1)\ntarget=rename_copy.T\ntarget.columns = target.iloc[0]\ntarget=target[1:]\ntarget.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T07:40:27.134007Z","iopub.execute_input":"2023-01-20T07:40:27.134392Z","iopub.status.idle":"2023-01-20T07:40:29.122604Z","shell.execute_reply.started":"2023-01-20T07:40:27.134342Z","shell.execute_reply":"2023-01-20T07:40:29.121250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def intersaction_coeff(list1, list2):\n    return len(set(list1).intersection(set(list2)))","metadata":{"execution":{"iopub.status.busy":"2023-01-20T07:40:29.125004Z","iopub.execute_input":"2023-01-20T07:40:29.125592Z","iopub.status.idle":"2023-01-20T07:40:29.134171Z","shell.execute_reply.started":"2023-01-20T07:40:29.125533Z","shell.execute_reply":"2023-01-20T07:40:29.132568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cos_sim(a, b):\n    cos_sim = np.dot(a, b)/(np.linalg.norm(a)*np.linalg.norm(b))\n    return cos_sim","metadata":{"execution":{"iopub.status.busy":"2023-01-20T07:40:29.139305Z","iopub.execute_input":"2023-01-20T07:40:29.139821Z","iopub.status.idle":"2023-01-20T07:40:29.148089Z","shell.execute_reply.started":"2023-01-20T07:40:29.139783Z","shell.execute_reply":"2023-01-20T07:40:29.146445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna=rna.T\nrna.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T07:40:29.150553Z","iopub.execute_input":"2023-01-20T07:40:29.150994Z","iopub.status.idle":"2023-01-20T07:40:29.205082Z","shell.execute_reply.started":"2023-01-20T07:40:29.150959Z","shell.execute_reply":"2023-01-20T07:40:29.203699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport catboost as cb\nimport gc\n\nX=rna\ny=target\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T07:40:29.207072Z","iopub.execute_input":"2023-01-20T07:40:29.207636Z","iopub.status.idle":"2023-01-20T07:40:41.901692Z","shell.execute_reply.started":"2023-01-20T07:40:29.207581Z","shell.execute_reply":"2023-01-20T07:40:41.900388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score, mean_squared_error\ndf_scores=pd.DataFrame()\nfor i in target.columns[20:30]:\n    print(i)\n    model = cb.CatBoostRegressor()\n    model.fit(X_train.loc[:,~X_train.columns.isin([i])], y_train[i])\n    model_dir_path = f\"/kaggle/working/models/{i}\"\n    os.makedirs(model_dir_path, exist_ok=True)\n\n    model_path = f\"/kaggle/working/models/{i}/model.cbm\"\n    model.save_model(model_path)\n    y_true=y_test[i]\n    y_pred = model.predict(X_test.loc[:,~X_test.columns.isin([i])])\n\n    s = r2_score(y_true,y_pred)\n    df_scores.loc['r2', i ]  = s\n    s = mean_squared_error(y_true,y_pred, squared=False) # squared=False -> RMSE, not MSE \n    df_scores.loc['RMSE', i ]  = s\n    print(f\"{i}, corr:\", df_scores)\n\n    del y_true\n    del y_pred\n    del model\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T07:40:41.903410Z","iopub.execute_input":"2023-01-20T07:40:41.903909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores.to_csv('df_scores_20-30.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}