{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport pickle as pkl\nfrom collections import defaultdict\nimport matplotlib.pyplot as plt\nimport catboost as cb\nimport lightgbm as lgb\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-20T10:16:36.408278Z","iopub.execute_input":"2023-01-20T10:16:36.408921Z","iopub.status.idle":"2023-01-20T10:16:38.693244Z","shell.execute_reply.started":"2023-01-20T10:16:36.408803Z","shell.execute_reply":"2023-01-20T10:16:38.692490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\nrename=pd.read_csv('/kaggle/input/research-project-01-around-multimodal-singlecell/CD_info.csv')\nrna.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T10:16:38.694658Z","iopub.execute_input":"2023-01-20T10:16:38.695083Z","iopub.status.idle":"2023-01-20T10:17:20.910522Z","shell.execute_reply.started":"2023-01-20T10:16:38.695059Z","shell.execute_reply":"2023-01-20T10:17:20.909435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna.columns=rna.columns.str.split('_').str[0]\nrna=rna.T\nrename_copy=rename.rename(columns={'ENSG ID':'gene_id'})\nrename_copy=rename_copy.merge(rna, on='gene_id', how='inner')\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T10:17:20.911810Z","iopub.execute_input":"2023-01-20T10:17:20.912165Z","iopub.status.idle":"2023-01-20T10:17:24.202662Z","shell.execute_reply.started":"2023-01-20T10:17:20.912130Z","shell.execute_reply":"2023-01-20T10:17:24.201216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rename_copy.shape","metadata":{"execution":{"iopub.status.busy":"2023-01-20T10:17:24.204213Z","iopub.execute_input":"2023-01-20T10:17:24.204524Z","iopub.status.idle":"2023-01-20T10:17:24.211308Z","shell.execute_reply.started":"2023-01-20T10:17:24.204499Z","shell.execute_reply":"2023-01-20T10:17:24.210141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rename_copy=rename_copy.drop(['Unnamed: 0','cd_names', 'names'], axis=1)\ntarget=rename_copy.T\ntarget.columns = target.iloc[0]\ntarget=target[1:]\ntarget.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T10:17:24.214945Z","iopub.execute_input":"2023-01-20T10:17:24.215390Z","iopub.status.idle":"2023-01-20T10:17:24.767838Z","shell.execute_reply.started":"2023-01-20T10:17:24.215351Z","shell.execute_reply":"2023-01-20T10:17:24.766853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def intersaction_coeff(list1, list2):\n    return len(set(list1).intersection(set(list2)))","metadata":{"execution":{"iopub.status.busy":"2023-01-20T10:17:24.768846Z","iopub.execute_input":"2023-01-20T10:17:24.769159Z","iopub.status.idle":"2023-01-20T10:17:24.774158Z","shell.execute_reply.started":"2023-01-20T10:17:24.769136Z","shell.execute_reply":"2023-01-20T10:17:24.772942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cos_sim(a, b):\n    cos_sim = np.dot(a, b)/(np.linalg.norm(a)*np.linalg.norm(b))\n    return cos_sim","metadata":{"execution":{"iopub.status.busy":"2023-01-20T10:17:24.775224Z","iopub.execute_input":"2023-01-20T10:17:24.775517Z","iopub.status.idle":"2023-01-20T10:17:24.786239Z","shell.execute_reply.started":"2023-01-20T10:17:24.775493Z","shell.execute_reply":"2023-01-20T10:17:24.785431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna=rna.T\nrna.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T10:17:24.787193Z","iopub.execute_input":"2023-01-20T10:17:24.788399Z","iopub.status.idle":"2023-01-20T10:17:24.832978Z","shell.execute_reply.started":"2023-01-20T10:17:24.788342Z","shell.execute_reply":"2023-01-20T10:17:24.831911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport catboost as cb\nimport gc\n\nX=rna\ny=target\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2023-01-20T10:17:24.836575Z","iopub.execute_input":"2023-01-20T10:17:24.836913Z","iopub.status.idle":"2023-01-20T10:17:32.188988Z","shell.execute_reply.started":"2023-01-20T10:17:24.836884Z","shell.execute_reply":"2023-01-20T10:17:32.187777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score, mean_squared_error\ndf_scores=pd.DataFrame()\nfor i in target.columns[101:110]:\n    print(i)\n    model = cb.CatBoostRegressor()\n    model.fit(X_train.loc[:,~X_train.columns.isin([i])], y_train[i])\n    model_dir_path = f\"/kaggle/working/models/{i}\"\n    os.makedirs(model_dir_path, exist_ok=True)\n\n    model_path = f\"/kaggle/working/models/{i}/model.cbm\"\n    model.save_model(model_path)\n    y_true=y_test[i]\n    y_pred = model.predict(X_test.loc[:,~X_test.columns.isin([i])])\n\n    s = r2_score(y_true,y_pred)\n    df_scores.loc['r2', i ]  = s\n    s = mean_squared_error(y_true,y_pred, squared=False) # squared=False -> RMSE, not MSE \n    df_scores.loc['RMSE', i ]  = s\n    print(f\"{i}, corr:\", df_scores)\n\n    del y_true\n    del y_pred\n    del model\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-20T10:17:32.190475Z","iopub.execute_input":"2023-01-20T10:17:32.190806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_scores.to_csv('df_scores_100-110.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}