{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport pickle as pkl\nfrom collections import defaultdict\nimport matplotlib.pyplot as plt\nimport catboost as cb\nimport lightgbm as lgb\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-20T07:59:04.415400Z","iopub.execute_input":"2023-01-20T07:59:04.416201Z","iopub.status.idle":"2023-01-20T07:59:07.049797Z","shell.execute_reply.started":"2023-01-20T07:59:04.416090Z","shell.execute_reply":"2023-01-20T07:59:07.048666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\nrename=pd.read_csv('/kaggle/input/research-project-01-around-multimodal-singlecell/CD_info.csv')\nrna.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T07:59:07.051452Z","iopub.execute_input":"2023-01-20T07:59:07.051786Z","iopub.status.idle":"2023-01-20T08:00:00.616155Z","shell.execute_reply.started":"2023-01-20T07:59:07.051754Z","shell.execute_reply":"2023-01-20T08:00:00.615033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna.columns=rna.columns.str.split('_').str[0]\nrna=rna.T\nrename_copy=rename.rename(columns={'ENSG ID':'gene_id'})\nrename_copy=rename_copy.merge(rna, on='gene_id', how='inner')\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:00:00.617479Z","iopub.execute_input":"2023-01-20T08:00:00.617816Z","iopub.status.idle":"2023-01-20T08:00:10.060729Z","shell.execute_reply.started":"2023-01-20T08:00:00.617786Z","shell.execute_reply":"2023-01-20T08:00:10.059075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rename_copy.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:00:10.063872Z","iopub.execute_input":"2023-01-20T08:00:10.064267Z","iopub.status.idle":"2023-01-20T08:00:10.072908Z","shell.execute_reply.started":"2023-01-20T08:00:10.064234Z","shell.execute_reply":"2023-01-20T08:00:10.071669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rename_copy=rename_copy.drop(['Unnamed: 0','cd_names', 'names'], axis=1)\ntarget=rename_copy.T\ntarget.columns = target.iloc[0]\ntarget=target[1:]\ntarget.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:00:10.074548Z","iopub.execute_input":"2023-01-20T08:00:10.075039Z","iopub.status.idle":"2023-01-20T08:00:12.879046Z","shell.execute_reply.started":"2023-01-20T08:00:10.074995Z","shell.execute_reply":"2023-01-20T08:00:12.874025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def intersaction_coeff(list1, list2):\n    return len(set(list1).intersection(set(list2)))","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:00:12.883395Z","iopub.execute_input":"2023-01-20T08:00:12.883803Z","iopub.status.idle":"2023-01-20T08:00:12.900816Z","shell.execute_reply.started":"2023-01-20T08:00:12.883769Z","shell.execute_reply":"2023-01-20T08:00:12.897036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cos_sim(a, b):\n    cos_sim = np.dot(a, b)/(np.linalg.norm(a)*np.linalg.norm(b))\n    return cos_sim","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:00:12.908564Z","iopub.execute_input":"2023-01-20T08:00:12.911472Z","iopub.status.idle":"2023-01-20T08:00:12.929917Z","shell.execute_reply.started":"2023-01-20T08:00:12.911109Z","shell.execute_reply":"2023-01-20T08:00:12.926618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna=rna.T\nrna.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:00:12.931042Z","iopub.execute_input":"2023-01-20T08:00:12.931461Z","iopub.status.idle":"2023-01-20T08:00:12.985447Z","shell.execute_reply.started":"2023-01-20T08:00:12.931410Z","shell.execute_reply":"2023-01-20T08:00:12.984214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport catboost as cb\nimport gc\n\nX=rna\ny=target\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:00:12.987302Z","iopub.execute_input":"2023-01-20T08:00:12.988640Z","iopub.status.idle":"2023-01-20T08:00:27.151862Z","shell.execute_reply.started":"2023-01-20T08:00:12.988593Z","shell.execute_reply":"2023-01-20T08:00:27.149659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score, mean_squared_error\ndf_scores=pd.DataFrame()\nfor i in target.columns[40:50]:\n    print(i)\n    model = cb.CatBoostRegressor()\n    model.fit(X_train.loc[:,~X_train.columns.isin([i])], y_train[i])\n    model_dir_path = f\"/kaggle/working/models/{i}\"\n    os.makedirs(model_dir_path, exist_ok=True)\n\n    model_path = f\"/kaggle/working/models/{i}/model.cbm\"\n    model.save_model(model_path)\n    y_true=y_test[i]\n    y_pred = model.predict(X_test.loc[:,~X_test.columns.isin([i])])\n\n    s = r2_score(y_true,y_pred)\n    df_scores.loc['r2', i ]  = s\n    s = mean_squared_error(y_true,y_pred, squared=False) # squared=False -> RMSE, not MSE \n    df_scores.loc['RMSE', i ]  = s\n    print(f\"{i}, corr:\", df_scores)\n\n    del y_true\n    del y_pred\n    del model\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T08:00:27.160138Z","iopub.execute_input":"2023-01-20T08:00:27.160602Z","iopub.status.idle":"2023-01-20T15:30:32.177999Z","shell.execute_reply.started":"2023-01-20T08:00:27.160555Z","shell.execute_reply":"2023-01-20T15:30:32.175013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores.to_csv('df_scores_40-50.csv')","metadata":{"execution":{"iopub.status.busy":"2023-01-20T15:30:32.180373Z","iopub.status.idle":"2023-01-20T15:30:32.180859Z","shell.execute_reply.started":"2023-01-20T15:30:32.180643Z","shell.execute_reply":"2023-01-20T15:30:32.180662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}