{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\n\nimport catboost as cb\nimport os\nimport gc\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import defaultdict\nfrom IPython.display import FileLink        \nimport pickle as pkl\n\nimport matplotlib.gridspec as gridspec\nimport lightgbm as lgmb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":false,"_kg_hide-output":false,"execution":{"iopub.status.busy":"2023-03-14T22:41:27.505455Z","iopub.execute_input":"2023-03-14T22:41:27.505866Z","iopub.status.idle":"2023-03-14T22:41:27.514515Z","shell.execute_reply.started":"2023-03-14T22:41:27.505834Z","shell.execute_reply":"2023-03-14T22:41:27.513006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Variables","metadata":{}},{"cell_type":"code","source":"RES_BASE_DIR = \"/kaggle/input/d/tttzof351/catboost-models-v7/models\"\nBASE_DIR = RES_BASE_DIR","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:41:28.891488Z","iopub.execute_input":"2023-03-14T22:41:28.891914Z","iopub.status.idle":"2023-03-14T22:41:28.897262Z","shell.execute_reply.started":"2023-03-14T22:41:28.891877Z","shell.execute_reply":"2023-03-14T22:41:28.895961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utils","metadata":{}},{"cell_type":"code","source":"def correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    \n    y2 = y_pred.copy()\n    y2 -= y2.mean(axis=0);    y2 /= y2.std(axis=0) \n    y1 = y_true.copy(); \n    y1 -= y1.mean(axis=0);    y1 /= y1.std(axis=0) \n        \n    c = (y1*y2).mean().mean()# Correlation for rescaled matrices is just matrix product and average \n        \n    return c","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-14T22:41:29.677170Z","iopub.execute_input":"2023-03-14T22:41:29.677569Z","iopub.status.idle":"2023-03-14T22:41:29.685833Z","shell.execute_reply.started":"2023-03-14T22:41:29.677537Z","shell.execute_reply":"2023-03-14T22:41:29.684428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalization(values):\n    return (values - np.mean(values)) / np.std(values)","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:41:30.000148Z","iopub.execute_input":"2023-03-14T22:41:30.000580Z","iopub.status.idle":"2023-03-14T22:41:30.005680Z","shell.execute_reply.started":"2023-03-14T22:41:30.000509Z","shell.execute_reply":"2023-03-14T22:41:30.004732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_pkl(obj, path):\n    with open(path, \"wb\") as file:\n        pkl.dump(obj, file)","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:41:30.304407Z","iopub.execute_input":"2023-03-14T22:41:30.305216Z","iopub.status.idle":"2023-03-14T22:41:30.311221Z","shell.execute_reply.started":"2023-03-14T22:41:30.305174Z","shell.execute_reply":"2023-03-14T22:41:30.310110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_pkl(path):\n    with open(path, \"rb\") as file:\n        return pkl.load(file)    \n    \nopen_pkl = read_pkl","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:41:30.560055Z","iopub.execute_input":"2023-03-14T22:41:30.560500Z","iopub.status.idle":"2023-03-14T22:41:30.566443Z","shell.execute_reply.started":"2023-03-14T22:41:30.560464Z","shell.execute_reply":"2023-03-14T22:41:30.564197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def intersaction_coeff(list1, list2):\n    return len(set(list1).intersection(set(list2)))","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:41:30.936540Z","iopub.execute_input":"2023-03-14T22:41:30.937674Z","iopub.status.idle":"2023-03-14T22:41:30.942177Z","shell.execute_reply.started":"2023-03-14T22:41:30.937613Z","shell.execute_reply":"2023-03-14T22:41:30.941222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cos_sim(a, b):\n    cos_sim = np.dot(a, b)/(np.linalg.norm(a)*np.linalg.norm(b))\n    return cos_sim","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:41:31.187679Z","iopub.execute_input":"2023-03-14T22:41:31.188109Z","iopub.status.idle":"2023-03-14T22:41:31.194406Z","shell.execute_reply.started":"2023-03-14T22:41:31.188074Z","shell.execute_reply":"2023-03-14T22:41:31.193106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LGBM all features","metadata":{}},{"cell_type":"code","source":"lgbm_df = pd.read_feather(f\"{RES_BASE_DIR}/fi_all_lgbm.feather\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:35:24.817725Z","iopub.execute_input":"2023-03-11T17:35:24.820916Z","iopub.status.idle":"2023-03-11T17:35:25.269400Z","shell.execute_reply.started":"2023-03-11T17:35:24.820848Z","shell.execute_reply":"2023-03-11T17:35:25.267970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:35:25.274888Z","iopub.execute_input":"2023-03-11T17:35:25.275737Z","iopub.status.idle":"2023-03-11T17:36:31.100543Z","shell.execute_reply.started":"2023-03-11T17:35:25.275686Z","shell.execute_reply":"2023-03-11T17:36:31.099466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rna_names = train_df.columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:31.101839Z","iopub.execute_input":"2023-03-11T17:36:31.102127Z","iopub.status.idle":"2023-03-11T17:36:31.107006Z","shell.execute_reply.started":"2023-03-11T17:36:31.102100Z","shell.execute_reply":"2023-03-11T17:36:31.105980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.xlabel(\"Experiment\")\nplt.ylabel(\"Corr\")\nplt.plot(np.sort(lgbm_df[\"corr\"]))","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:31.108141Z","iopub.execute_input":"2023-03-11T17:36:31.108401Z","iopub.status.idle":"2023-03-11T17:36:31.346401Z","shell.execute_reply.started":"2023-03-11T17:36:31.108373Z","shell.execute_reply":"2023-03-11T17:36:31.345334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_df[lgbm_df[\"corr\"] >= 0.7].CD.unique()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:31.347655Z","iopub.execute_input":"2023-03-11T17:36:31.347934Z","iopub.status.idle":"2023-03-11T17:36:31.357823Z","shell.execute_reply.started":"2023-03-11T17:36:31.347908Z","shell.execute_reply":"2023-03-11T17:36:31.356815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_high_corr_df = lgbm_df[lgbm_df[\"corr\"] >= 0.7]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:31.359185Z","iopub.execute_input":"2023-03-11T17:36:31.359998Z","iopub.status.idle":"2023-03-11T17:36:31.367177Z","shell.execute_reply.started":"2023-03-11T17:36:31.359967Z","shell.execute_reply":"2023-03-11T17:36:31.366272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def intersaction_coeff_k(fi_left, fi_right, k=20):\n    fi_left_best = np.argsort(fi_left)[::-1][:k]\n    fi_right_best = np.argsort(fi_right)[::-1][:k]\n\n    rna_left_best = np.array(rna_names)[fi_left_best]\n    rna_right_best = np.array(rna_names)[fi_right_best]\n\n    return intersaction_coeff(rna_left_best, rna_right_best)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:31.368638Z","iopub.execute_input":"2023-03-11T17:36:31.368905Z","iopub.status.idle":"2023-03-11T17:36:31.377792Z","shell.execute_reply.started":"2023-03-11T17:36:31.368880Z","shell.execute_reply":"2023-03-11T17:36:31.376923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"intersaction_coeff_k(\n    lgbm_high_corr_df.iloc[0].fi,\n    lgbm_high_corr_df.iloc[0].fi\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:31.379129Z","iopub.execute_input":"2023-03-11T17:36:31.379435Z","iopub.status.idle":"2023-03-11T17:36:31.404698Z","shell.execute_reply.started":"2023-03-11T17:36:31.379409Z","shell.execute_reply":"2023-03-11T17:36:31.403911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(np.sort(lgbm_high_corr_df.iloc[0].fi)[::-1][:2000])","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:31.405822Z","iopub.execute_input":"2023-03-11T17:36:31.406101Z","iopub.status.idle":"2023-03-11T17:36:31.554650Z","shell.execute_reply.started":"2023-03-11T17:36:31.406075Z","shell.execute_reply":"2023-03-11T17:36:31.553879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install distfit","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:31.555677Z","iopub.execute_input":"2023-03-11T17:36:31.556320Z","iopub.status.idle":"2023-03-11T17:36:44.887418Z","shell.execute_reply.started":"2023-03-11T17:36:31.556288Z","shell.execute_reply":"2023-03-11T17:36:44.884739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from distfit import distfit","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:44.891566Z","iopub.execute_input":"2023-03-11T17:36:44.892147Z","iopub.status.idle":"2023-03-11T17:36:45.798276Z","shell.execute_reply.started":"2023-03-11T17:36:44.892104Z","shell.execute_reply":"2023-03-11T17:36:45.797118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfit = distfit()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:45.800149Z","iopub.execute_input":"2023-03-11T17:36:45.800593Z","iopub.status.idle":"2023-03-11T17:36:45.807657Z","shell.execute_reply.started":"2023-03-11T17:36:45.800549Z","shell.execute_reply":"2023-03-11T17:36:45.806420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfit.fit_transform(lgbm_high_corr_df.iloc[0].fi)\ndfit.plot()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:45.809116Z","iopub.execute_input":"2023-03-11T17:36:45.810067Z","iopub.status.idle":"2023-03-11T17:36:50.478255Z","shell.execute_reply.started":"2023-03-11T17:36:45.810024Z","shell.execute_reply":"2023-03-11T17:36:50.477191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfit.model","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:50.479720Z","iopub.execute_input":"2023-03-11T17:36:50.480056Z","iopub.status.idle":"2023-03-11T17:36:50.488797Z","shell.execute_reply.started":"2023-03-11T17:36:50.480025Z","shell.execute_reply":"2023-03-11T17:36:50.487475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = dfit.plot(chart='PDF')\nfig, ax = dfit.plot(chart='CDF')","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:50.490699Z","iopub.execute_input":"2023-03-11T17:36:50.491591Z","iopub.status.idle":"2023-03-11T17:36:51.320205Z","shell.execute_reply.started":"2023-03-11T17:36:50.491547Z","shell.execute_reply":"2023-03-11T17:36:51.319173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = dfit.plot(chart='PDF', n_top=11)\nfig, ax = dfit.plot(chart='CDF', n_top=11)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:51.321732Z","iopub.execute_input":"2023-03-11T17:36:51.322073Z","iopub.status.idle":"2023-03-11T17:36:52.553017Z","shell.execute_reply.started":"2023-03-11T17:36:51.322041Z","shell.execute_reply":"2023-03-11T17:36:52.551909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = dfit.qqplot(lgbm_high_corr_df.iloc[0].fi)\nfig, ax = dfit.qqplot(lgbm_high_corr_df.iloc[0].fi, n_top=11)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:52.554646Z","iopub.execute_input":"2023-03-11T17:36:52.555565Z","iopub.status.idle":"2023-03-11T17:36:54.943039Z","shell.execute_reply.started":"2023-03-11T17:36:52.555517Z","shell.execute_reply":"2023-03-11T17:36:54.941869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(np.sort(lgbm_df[lgbm_df[\"CD\"] == \"CD36\"][\"corr\"]))","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:54.944778Z","iopub.execute_input":"2023-03-11T17:36:54.945432Z","iopub.status.idle":"2023-03-11T17:36:55.108508Z","shell.execute_reply.started":"2023-03-11T17:36:54.945367Z","shell.execute_reply":"2023-03-11T17:36:55.107740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_cd36_df = lgbm_df[lgbm_df[\"CD\"] == \"CD36\"]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:55.109731Z","iopub.execute_input":"2023-03-11T17:36:55.110497Z","iopub.status.idle":"2023-03-11T17:36:55.116289Z","shell.execute_reply.started":"2023-03-11T17:36:55.110462Z","shell.execute_reply":"2023-03-11T17:36:55.115014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lgbm_high_df.CD.unique().tolist()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:55.125694Z","iopub.execute_input":"2023-03-11T17:36:55.126008Z","iopub.status.idle":"2023-03-11T17:36:55.130447Z","shell.execute_reply.started":"2023-03-11T17:36:55.125979Z","shell.execute_reply":"2023-03-11T17:36:55.128982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## OLD FI","metadata":{}},{"cell_type":"code","source":"catboost_fi = open_pkl(f\"{RES_BASE_DIR}/catboost_feature_importances_v5.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:55.132213Z","iopub.execute_input":"2023-03-11T17:36:55.132870Z","iopub.status.idle":"2023-03-11T17:36:55.198995Z","shell.execute_reply.started":"2023-03-11T17:36:55.132836Z","shell.execute_reply":"2023-03-11T17:36:55.197936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catboost_fi.keys()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:55.200842Z","iopub.execute_input":"2023-03-11T17:36:55.201285Z","iopub.status.idle":"2023-03-11T17:36:55.208184Z","shell.execute_reply.started":"2023-03-11T17:36:55.201245Z","shell.execute_reply":"2023-03-11T17:36:55.207362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd41_fi = catboost_fi[\"CD41\"][\"best_100\"]\ncd36_fi = catboost_fi[\"CD36\"][\"best_100\"]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:55.209536Z","iopub.execute_input":"2023-03-11T17:36:55.210115Z","iopub.status.idle":"2023-03-11T17:36:55.216412Z","shell.execute_reply.started":"2023-03-11T17:36:55.210084Z","shell.execute_reply":"2023-03-11T17:36:55.215509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"intersaction_coeff(cd41_fi, cd36_fi)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:55.217490Z","iopub.execute_input":"2023-03-11T17:36:55.218668Z","iopub.status.idle":"2023-03-11T17:36:55.228723Z","shell.execute_reply.started":"2023-03-11T17:36:55.218586Z","shell.execute_reply":"2023-03-11T17:36:55.227787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = catboost_fi[\"CD41\"][\"scores\"]\nscores = np.flip(np.sort(scores))\nplt.plot(scores[:100])","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:36:55.230634Z","iopub.execute_input":"2023-03-11T17:36:55.231338Z","iopub.status.idle":"2023-03-11T17:36:55.371528Z","shell.execute_reply.started":"2023-03-11T17:36:55.231285Z","shell.execute_reply":"2023-03-11T17:36:55.370252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sec1","metadata":{}},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"code","source":"train_cite_input_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\ntrain_cite_target_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_targets.h5\")\nmetadata_df = pd.read_csv(\"/kaggle/input/open-problems-multimodal/metadata.csv\")\n\nday_4_cell_ids = set(metadata_df[metadata_df[\"day\"] == 4][\"cell_id\"])\n\nday_4_test_index = train_cite_input_df.index[train_cite_input_df.index.isin(day_4_cell_ids)]\ntrain_index = train_cite_input_df.index[~train_cite_input_df.index.isin(day_4_cell_ids)]","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:36:55.373160Z","iopub.execute_input":"2023-03-11T17:36:55.374466Z","iopub.status.idle":"2023-03-11T17:37:55.120143Z","shell.execute_reply.started":"2023-03-11T17:36:55.374405Z","shell.execute_reply":"2023-03-11T17:37:55.118893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df.set_index(\"cell_id\", inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:37:55.121678Z","iopub.execute_input":"2023-03-11T17:37:55.122149Z","iopub.status.idle":"2023-03-11T17:37:55.128749Z","shell.execute_reply.started":"2023-03-11T17:37:55.122104Z","shell.execute_reply":"2023-03-11T17:37:55.127545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_target_df = train_cite_target_df.join(metadata_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:37:55.130490Z","iopub.execute_input":"2023-03-11T17:37:55.131222Z","iopub.status.idle":"2023-03-11T17:37:55.286123Z","shell.execute_reply.started":"2023-03-11T17:37:55.131174Z","shell.execute_reply":"2023-03-11T17:37:55.285149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train catboosts","metadata":{}},{"cell_type":"code","source":"target_cds = [\n    \"CD44\",\n    \"CD41\",\n    \"CD36\",\n    \"CD32\",\n    \"CD88\",\n    \"CD48\",\n    \"CD62L\",\n    \"CD49b\",\n    \"CD45RA\",\n    \"CD82\",\n    \"CD38\",\n    \"CD71\",\n    \"CD115\",\n    \"CD11a\",\n    \"CD244\",\n    \"CD45\"    \n]\n\n# for cd_it in target_cds:\n\n#     model = cb.CatBoostRegressor(\n#         task_type=\"GPU\",\n#         loss_function=\"RMSE\",\n#         metric_period=100,\n#         verbose=1\n#     )\n\n#     model.fit(\n#         train_cite_input_df.loc[train_index], \n#         train_cite_target_df.loc[train_index][cd_it]\n#     )\n\n#     model_dir_path = f\"/kaggle/working/models/{cd_it}\"\n#     os.makedirs(model_dir_path, exist_ok=True)\n\n#     model_path = f\"/kaggle/working/models/{cd_it}/model.cbm\"\n#     model.save_model(model_path)\n\n#     y_true = train_cite_target_df.loc[day_4_test_index][cd_it].values\n\n#     y_pred = model.predict(\n#         train_cite_input_df.loc[day_4_test_index]\n#     )\n\n#     corr_score = correlation_score(y_true, y_pred)\n#     print(f\"{cd_it}, corr:\", corr_score)\n\n#     del y_true\n#     del y_pred\n#     del model\n#     gc.collect()","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:37:55.287832Z","iopub.execute_input":"2023-03-11T17:37:55.288584Z","iopub.status.idle":"2023-03-11T17:37:55.304769Z","shell.execute_reply.started":"2023-03-11T17:37:55.288535Z","shell.execute_reply":"2023-03-11T17:37:55.303165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!du -h /kaggle/working/models","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:37:55.306714Z","iopub.execute_input":"2023-03-11T17:37:55.307208Z","iopub.status.idle":"2023-03-11T17:37:55.316233Z","shell.execute_reply.started":"2023-03-11T17:37:55.307165Z","shell.execute_reply":"2023-03-11T17:37:55.315367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!zip -r /kaggle/working/models.zip /kaggle/working/models","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:37:55.317814Z","iopub.execute_input":"2023-03-11T17:37:55.318804Z","iopub.status.idle":"2023-03-11T17:37:55.328942Z","shell.execute_reply.started":"2023-03-11T17:37:55.318756Z","shell.execute_reply":"2023-03-11T17:37:55.327804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#FileLink(\"models.zip\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:37:55.330653Z","iopub.execute_input":"2023-03-11T17:37:55.331340Z","iopub.status.idle":"2023-03-11T17:37:55.339727Z","shell.execute_reply.started":"2023-03-11T17:37:55.331293Z","shell.execute_reply":"2023-03-11T17:37:55.338418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature importances","metadata":{}},{"cell_type":"code","source":"N = 100\nbest_features = defaultdict(list)\nall_best_features_dict = {}\n\nfor cd_it in tqdm(target_cds):\n    model_path = f\"{RES_BASE_DIR}/{cd_it}/model.cbm\"\n    model = cb.CatBoostRegressor()\n    model.load_model(model_path)\n\n    feature_importance_all = model.get_feature_importance()\n    max_indexes = np.flip(np.argsort(feature_importance_all))[:N]\n\n    feature_importance = feature_importance_all[max_indexes]\n    columns = train_cite_input_df.columns.tolist()\n    feature_names = train_cite_input_df.columns[max_indexes].tolist()\n    \n    for f_num in range(N):\n        best_features[f\"f_{f_num}\"].append(feature_names[f_num])\n        best_features[f\"f_{f_num}_score\"].append(feature_importance[f_num])        \n        \n    y_true = train_cite_target_df.loc[day_4_test_index][cd_it].values\n    y_pred = model.predict(\n        train_cite_input_df.loc[day_4_test_index]\n    )\n\n    corr_score = correlation_score(y_true, y_pred)            \n    \n    all_best_features_dict[cd_it] = {\n        \"scores\": feature_importance_all,\n        \"columns\": columns,\n        \"corr_score\": corr_score,\n        f\"best_{N}\": feature_names\n    }","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:37:55.341724Z","iopub.execute_input":"2023-03-11T17:37:55.342910Z","iopub.status.idle":"2023-03-11T17:39:17.427926Z","shell.execute_reply.started":"2023-03-11T17:37:55.342860Z","shell.execute_reply":"2023-03-11T17:39:17.426817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_best_features_dict = read_pkl(f\"{BASE_DIR}/catboost_feature_importances_v5.pkl\")\n#save_pkl(all_best_features_dict, \"./catboost_feature_importances_v5.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.429447Z","iopub.execute_input":"2023-03-11T17:39:17.430248Z","iopub.status.idle":"2023-03-11T17:39:17.457879Z","shell.execute_reply.started":"2023-03-11T17:39:17.430212Z","shell.execute_reply":"2023-03-11T17:39:17.456684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FileLink(\"catboost_feature_importances_v5.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.459714Z","iopub.execute_input":"2023-03-11T17:39:17.460149Z","iopub.status.idle":"2023-03-11T17:39:17.467909Z","shell.execute_reply.started":"2023-03-11T17:39:17.460107Z","shell.execute_reply":"2023-03-11T17:39:17.466685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_best_features_dict[\"CD36\"][\"columns\"][:10]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.469401Z","iopub.execute_input":"2023-03-11T17:39:17.469936Z","iopub.status.idle":"2023-03-11T17:39:17.480565Z","shell.execute_reply.started":"2023-03-11T17:39:17.469901Z","shell.execute_reply":"2023-03-11T17:39:17.479641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_cds = [\n    \"CD44\",\n    \"CD41\",\n    \"CD36\",\n    \"CD32\",\n    \"CD88\",\n    \"CD48\",\n    \"CD62L\",\n    \"CD49b\",\n    \"CD45RA\",\n    \"CD82\",\n    \"CD38\",\n    \"CD71\",\n    \"CD115\",\n    \"CD11a\",\n    \"CD244\",\n    \"CD45\"    \n]\n\n# N = 100\n# lgbm_best_features = defaultdict(list)\n# lgbm_all_best_features_dict = {}\n# for cd_it in tqdm(target_cds):\n#     model_path = f\"{RES_BASE_DIR}/lgmb_{cd_it}/model.txt\"\n#     if not os.path.exists(model_path):        \n#         print(f\"Don't have model: {cd_it}\")\n#         continue\n        \n#     model = lgmb.Booster(model_file=model_path)\n\n#     feature_importance_all = model.feature_importance()\n#     max_indexes = np.flip(np.argsort(feature_importance_all))[:N]\n\n#     feature_importance = feature_importance_all[max_indexes]\n#     columns = train_cite_input_df.columns.tolist()\n#     feature_names = train_cite_input_df.columns[max_indexes].tolist()\n    \n#     for f_num in range(N):\n#         lgbm_best_features[f\"f_{f_num}\"].append(feature_names[f_num])\n#         lgbm_best_features[f\"f_{f_num}_score\"].append(feature_importance[f_num])\n\n\n#     y_true = train_cite_target_df.loc[day_4_test_index][cd_it].values\n#     y_pred = model.predict(\n#         train_cite_input_df.loc[day_4_test_index]\n#     )\n\n#     corr_score = correlation_score(y_true, y_pred)            \n    \n#     lgbm_all_best_features_dict[cd_it] = {\n#         \"scores\": feature_importance_all,\n#         \"columns\": columns,\n#         \"corr_score\": corr_score,\n#         f\"best_{N}\": feature_names\n#     }    ","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.482530Z","iopub.execute_input":"2023-03-11T17:39:17.483023Z","iopub.status.idle":"2023-03-11T17:39:17.492854Z","shell.execute_reply.started":"2023-03-11T17:39:17.482979Z","shell.execute_reply":"2023-03-11T17:39:17.491704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_all_best_features_dict = open_pkl(f\"{BASE_DIR}/lgbm_feature_importances_v5.pkl\")\n#save_pkl(lgbm_all_best_features_dict, \"./lgbm_feature_importances_v5.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.495147Z","iopub.execute_input":"2023-03-11T17:39:17.495623Z","iopub.status.idle":"2023-03-11T17:39:17.616213Z","shell.execute_reply.started":"2023-03-11T17:39:17.495566Z","shell.execute_reply":"2023-03-11T17:39:17.615311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FileLink(\"lgbm_feature_importances_v5.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.617881Z","iopub.execute_input":"2023-03-11T17:39:17.618249Z","iopub.status.idle":"2023-03-11T17:39:17.622321Z","shell.execute_reply.started":"2023-03-11T17:39:17.618216Z","shell.execute_reply":"2023-03-11T17:39:17.621157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_cite_target_df[\"donor\"].unique())\nprint(train_cite_target_df[\"day\"].unique())","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.623792Z","iopub.execute_input":"2023-03-11T17:39:17.624217Z","iopub.status.idle":"2023-03-11T17:39:17.637687Z","shell.execute_reply.started":"2023-03-11T17:39:17.624175Z","shell.execute_reply":"2023-03-11T17:39:17.636696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = 100\ncv_all_best_features_dict = {}\n\ncd_it = \"CD36\"\n# for day_it in tqdm([2, 3, 4]):\n#     for donor_it in [32606, 13176, 31800]:\n#         model_path = f\"{RES_BASE_DIR}/cv_{cd_it}_day_{day_it}_donor_{donor_it}/model.cbm\"\n#         model = cb.CatBoostRegressor()\n#         model.load_model(model_path)\n        \n#         test_index = (train_cite_target_df[\"donor\"] == donor_it) & (train_cite_target_df[\"day\"] == day_it)\n        \n#         feature_importance_all = model.get_feature_importance()\n#         max_indexes = np.flip(np.argsort(feature_importance_all))[:N]\n\n#         feature_importance = feature_importance_all[max_indexes]\n#         columns = train_cite_input_df.columns.tolist()\n#         feature_names = train_cite_input_df.columns[max_indexes].tolist()\n\n#         y_true = train_cite_target_df[test_index][cd_it].values\n#         y_pred = model.predict(\n#             train_cite_input_df[test_index]\n#         )\n\n#         corr_score = correlation_score(y_true, y_pred)            \n\n#         cv_all_best_features_dict[f\"{cd_it}_{day_it}_{donor_it}\"] = {\n#             \"scores\": feature_importance_all,\n#             \"columns\": columns,\n#             \"corr_score\": corr_score,\n#             f\"best_{N}\": feature_names,\n#             \"cd\": cd_it,\n#             \"day\": day_it,\n#             \"donor\": donor_it\n#         }\n#         print(f\"{cd_it}_{day_it}_{donor_it}\",corr_score)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.639360Z","iopub.execute_input":"2023-03-11T17:39:17.640657Z","iopub.status.idle":"2023-03-11T17:39:17.648304Z","shell.execute_reply.started":"2023-03-11T17:39:17.640592Z","shell.execute_reply":"2023-03-11T17:39:17.647040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(cv_all_best_features_dict)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.649782Z","iopub.execute_input":"2023-03-11T17:39:17.650115Z","iopub.status.idle":"2023-03-11T17:39:17.662135Z","shell.execute_reply.started":"2023-03-11T17:39:17.650085Z","shell.execute_reply":"2023-03-11T17:39:17.661052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_all_best_features_dict = open_pkl(f\"{BASE_DIR}/cv_feature_importances_v5.pkl\")\n# save_pkl(cv_all_best_features_dict, \"./cv_feature_importances_v5.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.664900Z","iopub.execute_input":"2023-03-11T17:39:17.666047Z","iopub.status.idle":"2023-03-11T17:39:17.712363Z","shell.execute_reply.started":"2023-03-11T17:39:17.665998Z","shell.execute_reply":"2023-03-11T17:39:17.711499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FileLink(\"cv_feature_importances_v5.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.714069Z","iopub.execute_input":"2023-03-11T17:39:17.714549Z","iopub.status.idle":"2023-03-11T17:39:17.719447Z","shell.execute_reply.started":"2023-03-11T17:39:17.714492Z","shell.execute_reply":"2023-03-11T17:39:17.718652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = 100\ncell_type_all_best_features_dict = {}\n\ncell_types = [\n    \"HSC\",\n    \"EryP\",\n    \"BP\",\n    \"MasP\",\n    \"MkP\",\n    \"Mop\",\n    \"NeuP\",\n    \"hidden\"\n]\n\ntarget_cds = [\n    \"CD36\",\n    \"CD38\",\n    \"CD41\",\n    \"CD45RA\",\n    \"CD48\",\n    \"CD49b\",\n    \"CD62\",\n    \"CD71\",\n    \"CD82\",\n    \"CD32\",\n    \"CD45\",\n    \"CD88\"\n]\n\n# for cd_it in tqdm(target_cds):\n#     for cell_type_it in cell_types:\n#         model_path = f\"{RES_BASE_DIR}/{cd_it}_{cell_type_it}/model.cbm\"        \n#         test_index = (train_cite_target_df[\"day\"] == 4) & (train_cite_target_df[\"cell_type\"] == cell_type_it)\n        \n#         if not os.path.exists(model_path):\n#             print(\"Not exists path\")\n#             continue\n            \n#         if len(test_index) == 0:\n#             print(\"Empty test\")\n#             continue      \n        \n#         model = cb.CatBoostRegressor()\n#         model.load_model(model_path)\n                \n#         feature_importance_all = model.get_feature_importance()\n#         max_indexes = np.flip(np.argsort(feature_importance_all))[:N]\n\n#         feature_importance = feature_importance_all[max_indexes]\n#         columns = train_cite_input_df.columns.tolist()\n#         feature_names = train_cite_input_df.columns[max_indexes].tolist()\n\n#         y_true = train_cite_target_df[test_index][cd_it].values\n#         y_pred = model.predict(\n#             train_cite_input_df[test_index]\n#         )\n\n#         corr_score = correlation_score(y_true, y_pred)            \n\n#         cell_type_all_best_features_dict[f\"{cd_it}_{cell_type_it}\"] = {\n#             \"scores\": feature_importance_all,\n#             \"columns\": columns,\n#             \"corr_score\": corr_score,\n#             f\"best_{N}\": feature_names,\n#             \"cd\": cd_it,\n#             \"cell_type\": cell_type_it\n#         }  \n#         print(f\"{cd_it}_{cell_type_it}\",corr_score)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.730895Z","iopub.execute_input":"2023-03-11T17:39:17.731682Z","iopub.status.idle":"2023-03-11T17:39:17.739740Z","shell.execute_reply.started":"2023-03-11T17:39:17.731627Z","shell.execute_reply":"2023-03-11T17:39:17.738563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(cell_type_all_best_features_dict)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.741153Z","iopub.execute_input":"2023-03-11T17:39:17.741688Z","iopub.status.idle":"2023-03-11T17:39:17.753010Z","shell.execute_reply.started":"2023-03-11T17:39:17.741651Z","shell.execute_reply":"2023-03-11T17:39:17.751522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cell_type_all_best_features_dict = open_pkl(f\"{BASE_DIR}/cell_type_feature_importances_v5.pkl\")\n# save_pkl(cell_type_all_best_features_dict, \"./cell_type_feature_importances_v5.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:17.754433Z","iopub.execute_input":"2023-03-11T17:39:17.754834Z","iopub.status.idle":"2023-03-11T17:39:18.123673Z","shell.execute_reply.started":"2023-03-11T17:39:17.754799Z","shell.execute_reply":"2023-03-11T17:39:18.122749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FileLink(\"cell_type_feature_importances_v5.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:18.127587Z","iopub.execute_input":"2023-03-11T17:39:18.127998Z","iopub.status.idle":"2023-03-11T17:39:18.133030Z","shell.execute_reply.started":"2023-03-11T17:39:18.127963Z","shell.execute_reply":"2023-03-11T17:39:18.131922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature importance table","metadata":{}},{"cell_type":"code","source":"# pd.set_option('display.max_columns', None)\n# corr_scores_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:18.134516Z","iopub.execute_input":"2023-03-11T17:39:18.135521Z","iopub.status.idle":"2023-03-11T17:39:18.141850Z","shell.execute_reply.started":"2023-03-11T17:39:18.135475Z","shell.execute_reply":"2023-03-11T17:39:18.140689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CD36 and something similar ","metadata":{}},{"cell_type":"code","source":"# fi_scores = np.array([all_best_features_dict[cd_it][\"scores\"] for cd_it in target_cds])\n# cd_36_fi = fi_scores[target_cds.index(\"CD36\")]\n# norms = np.linalg.norm(fi_scores, axis=1) * np.linalg.norm(cd_36_fi)\n# cos_similarity_idx = (np.dot(fi_scores, cd_36_fi) / norms)\n\n# max_idxs = np.flip(np.argsort(cos_similarity_idx))\n# corr_scores_df.insert(2, \"CD36_similarity\", cos_similarity_idx)\n# cd36_best = set(all_best_features_dict[\"CD36\"][\"best_100\"])\n\n# def intersaction_with_cd36(features):\n#     return len(cd36_best.intersection(set(features)))\n\n# intersections_coeff = [\n#     intersaction_with_cd36(all_best_features_dict[cd_it][\"best_100\"]) for cd_it in target_cds\n# ]\n\n# corr_scores_df.insert(3, \"CD36_inter\", intersections_coeff)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:18.143565Z","iopub.execute_input":"2023-03-11T17:39:18.144328Z","iopub.status.idle":"2023-03-11T17:39:18.150818Z","shell.execute_reply.started":"2023-03-11T17:39:18.144281Z","shell.execute_reply":"2023-03-11T17:39:18.150078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# corr_scores_df.sort_values(\"CD36_similarity\", ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:18.152344Z","iopub.execute_input":"2023-03-11T17:39:18.152870Z","iopub.status.idle":"2023-03-11T17:39:18.159503Z","shell.execute_reply.started":"2023-03-11T17:39:18.152828Z","shell.execute_reply":"2023-03-11T17:39:18.158627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature importances plots","metadata":{}},{"cell_type":"code","source":"def plot_features_importance(\n    all_best_features_dict,\n    cd_name: str\n):\n    fig, (ax1, ax2) = plt.subplots(1, 2)\n    fig.set_size_inches(18, 7)\n\n    plt.suptitle(cd_name, fontsize=14)\n\n    all_scores = all_best_features_dict[cd_name][\"scores\"]\n    \n    max_index = np.argmax(all_scores)\n    max_value = all_scores[max_index]\n    max_name = all_best_features_dict[cd_name][\"columns\"][max_index]\n    scores_non_zero = all_scores[all_scores != 0.0]\n    scores_non_zero = np.sort(scores_non_zero)\n\n    if max_index < 15000:\n        delta = 50\n    else:\n        delta = -7000\n\n    ax1.text(max_index + delta, max_value, max_name)\n    ax1.set_xlabel(\"feature index\")\n    ax1.set_ylabel(\"feature_importance\")\n\n    _ = ax1.plot(all_scores)\n    \n    ax2.set_xlabel(\"sorted feature index (for non zero feature importance)\")\n    ax2.set_ylabel(\"log_(feature_importance)\")\n    ax2.set_yscale('log')\n    _ = ax2.plot(scores_non_zero)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:18.161195Z","iopub.execute_input":"2023-03-11T17:39:18.161859Z","iopub.status.idle":"2023-03-11T17:39:18.183521Z","shell.execute_reply.started":"2023-03-11T17:39:18.161826Z","shell.execute_reply":"2023-03-11T17:39:18.182239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_features_importance(all_best_features_dict, \"CD36\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:18.185086Z","iopub.execute_input":"2023-03-11T17:39:18.185705Z","iopub.status.idle":"2023-03-11T17:39:18.802856Z","shell.execute_reply.started":"2023-03-11T17:39:18.185670Z","shell.execute_reply":"2023-03-11T17:39:18.801654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_features_importance(all_best_features_dict, \"CD45\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:18.804436Z","iopub.execute_input":"2023-03-11T17:39:18.804948Z","iopub.status.idle":"2023-03-11T17:39:19.647249Z","shell.execute_reply.started":"2023-03-11T17:39:18.804904Z","shell.execute_reply":"2023-03-11T17:39:19.646082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# corr_scores_df.to_csv(\"/kaggle/working/corr_scores_df.csv\", index=False)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:19.649171Z","iopub.execute_input":"2023-03-11T17:39:19.649840Z","iopub.status.idle":"2023-03-11T17:39:19.654253Z","shell.execute_reply.started":"2023-03-11T17:39:19.649804Z","shell.execute_reply":"2023-03-11T17:39:19.653356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Intersaction between feature importances vs Corr confusion matrix","metadata":{}},{"cell_type":"code","source":"target_cds = [\n    \"CD44\",\n    \"CD41\",\n    \"CD36\",\n    \"CD32\",\n    \"CD88\",\n    \"CD48\",\n    \"CD62L\",\n    \"CD49b\",\n    \"CD45RA\",\n    \"CD82\",\n    \"CD38\",\n    \"CD71\",\n    \"CD115\",\n    \"CD11a\",\n    \"CD244\",\n    \"CD45\"    \n]\n\ncolumns_coeffies_fi = defaultdict(list)\nfor column_cd_it in tqdm(target_cds):\n    for row_cd_it in target_cds:\n        row_set = set(all_best_features_dict[row_cd_it][\"best_100\"])\n        column_set = set(all_best_features_dict[column_cd_it][\"best_100\"])\n        intersection_coeff = len(row_set.intersection(column_set))\n        \n        columns_coeffies_fi[column_cd_it].append(intersection_coeff)\n\ncolumns_coeffies_fi_df = pd.DataFrame(columns_coeffies_fi, index=target_cds)\n\n\ncolumns_coeffies_corr = defaultdict(list)\nfor column_cd_it in tqdm(target_cds):\n    for row_cd_it in target_cds:\n        row_values = train_cite_target_df.loc[train_index][row_cd_it]\n        column_values = train_cite_target_df.loc[train_index][column_cd_it]\n        \n        corr_coeff = correlation_score(row_values, column_values)\n        \n        columns_coeffies_corr[column_cd_it].append(corr_coeff)\n\ncolumns_coeffies_corr_df = pd.DataFrame(columns_coeffies_corr, index=target_cds)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:19.655729Z","iopub.execute_input":"2023-03-11T17:39:19.656029Z","iopub.status.idle":"2023-03-11T17:39:42.428940Z","shell.execute_reply.started":"2023-03-11T17:39:19.656001Z","shell.execute_reply":"2023-03-11T17:39:42.427557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plot heatmaps","metadata":{}},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1,2)\n#plt.figure(figsize=(40,8))\nfig.set_size_inches(18, 7)\n\nax1.set_title(\"Feature importance intersection\")\n_ = sns.heatmap(\n    columns_coeffies_fi_df, \n    xticklabels=columns_coeffies_fi_df.columns.values,\n    yticklabels=columns_coeffies_fi_df.columns.values,\n    ax=ax1,\n    cmap=\"Greens\"\n)\n\nax2.set_title(\"Corr confusions\")\n_ = sns.heatmap(\n    columns_coeffies_corr_df, \n    xticklabels=columns_coeffies_corr_df.columns.values,\n    yticklabels=columns_coeffies_corr_df.columns.values,\n    ax=ax2,\n    cmap=\"Greens\"    \n)\nplt.show()","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:42.430410Z","iopub.execute_input":"2023-03-11T17:39:42.431168Z","iopub.status.idle":"2023-03-11T17:39:43.221351Z","shell.execute_reply.started":"2023-03-11T17:39:42.431129Z","shell.execute_reply":"2023-03-11T17:39:43.219925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train for cell_types","metadata":{}},{"cell_type":"code","source":"cell_types = [\n    \"HSC\",\n    \"EryP\",\n    \"BP\",\n    \"MasP\",\n    \"MkP\",\n    \"Mop\",\n    \"NeuP\",\n    \"hidden\"\n]","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:43.223715Z","iopub.execute_input":"2023-03-11T17:39:43.224224Z","iopub.status.idle":"2023-03-11T17:39:43.230262Z","shell.execute_reply.started":"2023-03-11T17:39:43.224174Z","shell.execute_reply":"2023-03-11T17:39:43.228857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_cds = [\n#     \"CD36\",\n#     \"CD41\",\n#     \"CD32\",\n#     \"CD45\",\n#     \"CD88\",\n#     \"CD48\",\n#     \"CD62L\",\n#     \"CD49b\",\n    \"CD45RA\",\n    \"CD82\",\n    \"CD38\",\n    \"CD71\",        \n#     \"CD88\",\n#     \"CD48\",\n#     \"CD62L\",\n#     \"CD49b\",\n#     \"CD115\",\n#     \"CD11a\",\n#     \"CD244\",\n]","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:43.231910Z","iopub.execute_input":"2023-03-11T17:39:43.232648Z","iopub.status.idle":"2023-03-11T17:39:43.243271Z","shell.execute_reply.started":"2023-03-11T17:39:43.232574Z","shell.execute_reply":"2023-03-11T17:39:43.242031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for cd_it in target_cds:\n#     for cell_type_it in cell_types:\n#         model_dir_path = f\"/kaggle/working/models/{cd_it}_{cell_type_it}\"\n#         if os.path.exists(model_dir_path):\n#             print(\"Already exist\")\n#             continue\n            \n        \n#         set_cell_type_it = set(metadata_df[metadata_df[\"cell_type\"] == cell_type_it][\"cell_id\"].tolist())\n#         cell_type_test_index = day_4_test_index[day_4_test_index.isin(set_cell_type_it)]\n#         cell_type_train_index = train_index[train_index.isin(set_cell_type_it)]    \n        \n#         del set_cell_type_it\n        \n#         if len(cell_type_test_index) == 0 or len(cell_type_train_index) == 0:\n#             print(\"Empty. test:\", len(cell_type_test_index), \"; train:\", len(cell_type_train_index))\n#             continue\n\n#         model = cb.CatBoostRegressor(\n#             task_type=\"GPU\",\n#             loss_function=\"RMSE\",\n#             metric_period=100,\n#             verbose=1\n#         )\n\n#         model.fit(\n#             train_cite_input_df.loc[cell_type_train_index], \n#             train_cite_target_df.loc[cell_type_train_index][cd_it]\n#         )\n\n#         os.makedirs(model_dir_path, exist_ok=True)\n\n#         model_path = f\"/kaggle/working/models/{cd_it}_{cell_type_it}/model.cbm\"\n#         model.save_model(model_path)\n\n#         y_true = train_cite_target_df.loc[cell_type_test_index][cd_it].values\n\n#         y_pred = model.predict(\n#             train_cite_input_df.loc[cell_type_test_index]\n#         )\n\n#         corr_score = correlation_score(y_true, y_pred)\n#         print(f\"{cd_it}_{cell_type_it}, corr:\", corr_score)\n\n#         del y_true\n#         del y_pred\n#         del model\n#         del cell_type_test_index\n#         del cell_type_train_index\n\n#         gc.collect()","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:43.244852Z","iopub.execute_input":"2023-03-11T17:39:43.245518Z","iopub.status.idle":"2023-03-11T17:39:43.256776Z","shell.execute_reply.started":"2023-03-11T17:39:43.245467Z","shell.execute_reply":"2023-03-11T17:39:43.255554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !zip -r /kaggle/working/models.zip /kaggle/working/models","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:43.258204Z","iopub.execute_input":"2023-03-11T17:39:43.258613Z","iopub.status.idle":"2023-03-11T17:39:43.272764Z","shell.execute_reply.started":"2023-03-11T17:39:43.258561Z","shell.execute_reply":"2023-03-11T17:39:43.271446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !du -h /kaggle/working/models.zip","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:43.275731Z","iopub.execute_input":"2023-03-11T17:39:43.276302Z","iopub.status.idle":"2023-03-11T17:39:43.284401Z","shell.execute_reply.started":"2023-03-11T17:39:43.276252Z","shell.execute_reply":"2023-03-11T17:39:43.283388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FileLink(r'models.zip')","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:43.285780Z","iopub.execute_input":"2023-03-11T17:39:43.287118Z","iopub.status.idle":"2023-03-11T17:39:43.299184Z","shell.execute_reply.started":"2023-03-11T17:39:43.287061Z","shell.execute_reply":"2023-03-11T17:39:43.297964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature importance for cell_types","metadata":{}},{"cell_type":"code","source":"metadata_df[\"cell_id\"] = metadata_df.index.tolist()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:39:43.300732Z","iopub.execute_input":"2023-03-11T17:39:43.301627Z","iopub.status.idle":"2023-03-11T17:39:43.330190Z","shell.execute_reply.started":"2023-03-11T17:39:43.301556Z","shell.execute_reply":"2023-03-11T17:39:43.328911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = 100\nthreshold_corr = 0.0\ninteresting_pairs = []\ncell_types_feature_importances = defaultdict(dict)\n\ncell_types = [\n    \"HSC\",\n    \"EryP\",\n    \"BP\",\n    \"MasP\",\n    \"MkP\",\n    \"Mop\",\n    \"NeuP\",\n    \"hidden\"\n]\n\ntarget_cds = [\n    \"CD36\",\n    \"CD38\",\n    \"CD41\",\n    \"CD45RA\",\n    \"CD48\",\n    \"CD49b\",\n    \"CD62\",\n    \"CD71\",\n    \"CD82\",\n    \"CD32\",\n    \"CD45\",\n    \"CD88\"\n]\n\ntest_train_lens = defaultdict(list)\n\nfor cell_type_it in tqdm(cell_types):\n    set_cell_type_it = set(metadata_df[metadata_df[\"cell_type\"] == cell_type_it][\"cell_id\"].tolist())\n    cell_type_test_index = day_4_test_index[day_4_test_index.isin(set_cell_type_it)]\n    cell_type_train_index = train_index[train_index.isin(set_cell_type_it)]    \n\n    test_train_lens[\"cell_type\"].append(cell_type_it)\n    test_train_lens[\"train_len\"].append(len(cell_type_train_index))\n    test_train_lens[\"test_len\"].append(len(cell_type_test_index))\n    \n    for cd_it in target_cds:\n        model_path = f\"{RES_BASE_DIR}/{cd_it}_{cell_type_it}/model.cbm\"\n\n        if os.path.exists(model_path):    \n            model = cb.CatBoostRegressor()\n            model.load_model(model_path)\n\n            feature_importance_all = model.get_feature_importance()\n            max_indexes = np.flip(np.argsort(feature_importance_all))[:N]\n\n            feature_importance = feature_importance_all[max_indexes]\n            feature_names = train_cite_input_df.columns[max_indexes].tolist()\n            \n            cell_types_feature_importances[cd_it][cell_type_it] = {\n                f\"best_{N}_names\": feature_names,\n                f\"best_{N}_values\": feature_importance\n            }\n            \n            y_true = train_cite_target_df.loc[cell_type_test_index][cd_it].values\n            y_pred = model.predict(train_cite_input_df.loc[cell_type_test_index])\n\n            corr_score = correlation_score(y_true, y_pred)\n            test_train_lens[f\"{cd_it}_corr\"].append(corr_score)\n            \n            if corr_score > threshold_corr:\n                interesting_pairs.append({\n                    \"cell_type\": cell_type_it,\n                    \"cd\": cd_it,\n                    \"corr\": corr_score                    \n                })\n        else:\n            test_train_lens[f\"{cd_it}_corr\"].append(None)\n                        \n    del set_cell_type_it\n    del cell_type_test_index\n    del cell_type_train_index\n    \n    gc.collect()\n    \ntest_train_lens_df = pd.DataFrame(test_train_lens)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:39:43.332013Z","iopub.execute_input":"2023-03-11T17:39:43.332377Z","iopub.status.idle":"2023-03-11T17:42:42.775649Z","shell.execute_reply.started":"2023-03-11T17:39:43.332345Z","shell.execute_reply":"2023-03-11T17:42:42.774482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_train_lens_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:42:42.777175Z","iopub.execute_input":"2023-03-11T17:42:42.777525Z","iopub.status.idle":"2023-03-11T17:42:42.803087Z","shell.execute_reply.started":"2023-03-11T17:42:42.777475Z","shell.execute_reply":"2023-03-11T17:42:42.802030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interesting_pairs_intersections = defaultdict(list)\n\nfor inter_it in interesting_pairs:\n    cell_type_it = inter_it[\"cell_type\"]\n    cd_it = inter_it[\"cd\"]\n    \n    cell_type_fi = set(cell_types_feature_importances[cd_it][cell_type_it][f\"best_{N}_names\"])\n    full_fi = set(all_best_features_dict[cd_it][f\"best_{N}\"])\n    intersection_coeff = len(full_fi.intersection(cell_type_fi))\n    \n    interesting_pairs_intersections[\"cd\"].append(cd_it)\n    interesting_pairs_intersections[\"cell_type\"].append(cell_type_it)\n    interesting_pairs_intersections[\"intersection\"].append(intersection_coeff)\n    \ninteresting_pairs_intersections_df = pd.DataFrame(interesting_pairs_intersections)\ninteresting_pairs_intersections_df.sort_values(\"intersection\", inplace=True)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:42:42.806981Z","iopub.execute_input":"2023-03-11T17:42:42.807329Z","iopub.status.idle":"2023-03-11T17:42:42.819574Z","shell.execute_reply.started":"2023-03-11T17:42:42.807297Z","shell.execute_reply":"2023-03-11T17:42:42.818181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd_it = \"CD36\"\ncell_type_it = \"MasP\"\ncell_type_fi = set(cell_types_feature_importances[cd_it][cell_type_it][f\"best_{N}_names\"])\nfull_fi = set(all_best_features_dict[cd_it][f\"best_{N}\"])\nintersection_coeff_names = full_fi.intersection(cell_type_fi)","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:42:42.821513Z","iopub.execute_input":"2023-03-11T17:42:42.822011Z","iopub.status.idle":"2023-03-11T17:42:42.830357Z","shell.execute_reply.started":"2023-03-11T17:42:42.821966Z","shell.execute_reply":"2023-03-11T17:42:42.829330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"intersection_coeff_names","metadata":{"_kg_hide-input":false,"execution":{"iopub.status.busy":"2023-03-11T17:42:42.831544Z","iopub.execute_input":"2023-03-11T17:42:42.831857Z","iopub.status.idle":"2023-03-11T17:42:42.842879Z","shell.execute_reply.started":"2023-03-11T17:42:42.831829Z","shell.execute_reply":"2023-03-11T17:42:42.841970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interesting_pairs_intersections_df[interesting_pairs_intersections_df[\"intersection\"] > 30]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:42:42.844190Z","iopub.execute_input":"2023-03-11T17:42:42.844497Z","iopub.status.idle":"2023-03-11T17:42:42.862525Z","shell.execute_reply.started":"2023-03-11T17:42:42.844460Z","shell.execute_reply":"2023-03-11T17:42:42.861678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sec2","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import catboost as cb\nimport os\nimport gc\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import defaultdict\nfrom IPython.display import FileLink        \n\n\nimport matplotlib.gridspec as gridspec","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:42:42.864163Z","iopub.execute_input":"2023-03-11T17:42:42.864492Z","iopub.status.idle":"2023-03-11T17:42:42.874539Z","shell.execute_reply.started":"2023-03-11T17:42:42.864463Z","shell.execute_reply":"2023-03-11T17:42:42.873555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U memoization","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:42:42.876047Z","iopub.execute_input":"2023-03-11T17:42:42.876359Z","iopub.status.idle":"2023-03-11T17:42:56.349880Z","shell.execute_reply.started":"2023-03-11T17:42:42.876330Z","shell.execute_reply":"2023-03-11T17:42:56.348643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from memoization import cached","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:42:56.353126Z","iopub.execute_input":"2023-03-11T17:42:56.353476Z","iopub.status.idle":"2023-03-11T17:42:56.370627Z","shell.execute_reply.started":"2023-03-11T17:42:56.353443Z","shell.execute_reply":"2023-03-11T17:42:56.369695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"code","source":"del train_cite_input_df\ndel train_cite_target_df\ndel metadata_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:42:56.389939Z","iopub.execute_input":"2023-03-11T17:42:56.391063Z","iopub.status.idle":"2023-03-11T17:42:56.441468Z","shell.execute_reply.started":"2023-03-11T17:42:56.391007Z","shell.execute_reply":"2023-03-11T17:42:56.440388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:42:56.442672Z","iopub.execute_input":"2023-03-11T17:42:56.443574Z","iopub.status.idle":"2023-03-11T17:42:56.686506Z","shell.execute_reply.started":"2023-03-11T17:42:56.443537Z","shell.execute_reply":"2023-03-11T17:42:56.684864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_inputs_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\ntrain_cite_targets_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_targets.h5\")\nmetadata_df = pd.read_csv(\"/kaggle/input/open-problems-multimodal/metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:42:56.688325Z","iopub.execute_input":"2023-03-11T17:42:56.689218Z","iopub.status.idle":"2023-03-11T17:43:52.432801Z","shell.execute_reply.started":"2023-03-11T17:42:56.689180Z","shell.execute_reply":"2023-03-11T17:43:52.431657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df = metadata_df.set_index(\"cell_id\")\n\ntrain_cite_targets_df = train_cite_targets_df.join(metadata_df)\ntrain_cite_inputs_df = train_cite_inputs_df.join(metadata_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:52.434165Z","iopub.execute_input":"2023-03-11T17:43:52.434491Z","iopub.status.idle":"2023-03-11T17:43:56.718528Z","shell.execute_reply.started":"2023-03-11T17:43:52.434461Z","shell.execute_reply":"2023-03-11T17:43:56.717575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utils","metadata":{}},{"cell_type":"code","source":"class SeabornFig2Grid():\n\n    def __init__(self, seaborngrid, fig,  subplot_spec):\n        self.fig = fig\n        self.sg = seaborngrid\n        self.subplot = subplot_spec\n        if isinstance(self.sg, sns.axisgrid.FacetGrid) or \\\n            isinstance(self.sg, sns.axisgrid.PairGrid):\n            self._movegrid()\n        elif isinstance(self.sg, sns.axisgrid.JointGrid):\n            self._movejointgrid()\n        self._finalize()\n\n    def _movegrid(self):\n        \"\"\" Move PairGrid or Facetgrid \"\"\"\n        self._resize()\n        n = self.sg.axes.shape[0]\n        m = self.sg.axes.shape[1]\n        self.subgrid = gridspec.GridSpecFromSubplotSpec(n,m, subplot_spec=self.subplot)\n        for i in range(n):\n            for j in range(m):\n                self._moveaxes(self.sg.axes[i,j], self.subgrid[i,j])\n\n    def _movejointgrid(self):\n        \"\"\" Move Jointgrid \"\"\"\n        h= self.sg.ax_joint.get_position().height\n        h2= self.sg.ax_marg_x.get_position().height\n        r = int(np.round(h/h2))\n        self._resize()\n        self.subgrid = gridspec.GridSpecFromSubplotSpec(r+1,r+1, subplot_spec=self.subplot)\n\n        self._moveaxes(self.sg.ax_joint, self.subgrid[1:, :-1])\n        self._moveaxes(self.sg.ax_marg_x, self.subgrid[0, :-1])\n        self._moveaxes(self.sg.ax_marg_y, self.subgrid[1:, -1])\n\n    def _moveaxes(self, ax, gs):\n        #https://stackoverflow.com/a/46906599/4124317\n        ax.remove()\n        ax.figure=self.fig\n        self.fig.axes.append(ax)\n        self.fig.add_axes(ax)\n        ax._subplotspec = gs\n        ax.set_position(gs.get_position(self.fig))\n        ax.set_subplotspec(gs)\n\n    def _finalize(self):\n        plt.close(self.sg.fig)\n        self.fig.canvas.mpl_connect(\"resize_event\", self._resize)\n        self.fig.canvas.draw()\n\n    def _resize(self, evt=None):\n        self.sg.fig.set_size_inches(self.fig.get_size_inches())","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:56.719912Z","iopub.execute_input":"2023-03-11T17:43:56.720320Z","iopub.status.idle":"2023-03-11T17:43:56.737654Z","shell.execute_reply.started":"2023-03-11T17:43:56.720281Z","shell.execute_reply":"2023-03-11T17:43:56.736898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@cached\ndef get_count_df(\n    donor_it: int = 32606,\n    cd_it: str = \"CD36\"\n):\n    count_df = train_cite_targets_df[\n        train_cite_targets_df[\"donor\"] == donor_it\n    ][[cd_it, \"day\"]].groupby([\"day\"]).count()\n\n    count_df[count_df.index.name] = count_df.index\n    \n    return count_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:56.738942Z","iopub.execute_input":"2023-03-11T17:43:56.739493Z","iopub.status.idle":"2023-03-11T17:43:56.752401Z","shell.execute_reply.started":"2023-03-11T17:43:56.739460Z","shell.execute_reply":"2023-03-11T17:43:56.751452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@cached\ndef get_mean_cd_df(\n    donor_it: int = 32606,\n    cd_it: str = \"CD36\"\n):\n    mean_df = per_day_df = train_cite_targets_df[\n        train_cite_targets_df[\"donor\"] == donor_it\n    ][[cd_it, \"day\"]] \\\n    .groupby([\"day\"]) \\\n    .mean() \\\n    .reset_index()\n    \n    return mean_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:56.753901Z","iopub.execute_input":"2023-03-11T17:43:56.754251Z","iopub.status.idle":"2023-03-11T17:43:56.765585Z","shell.execute_reply.started":"2023-03-11T17:43:56.754220Z","shell.execute_reply":"2023-03-11T17:43:56.764809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@cached\ndef get_per_cell_type_count(\n    donor_it: int = 32606,\n    cd_it: str = \"CD36\"\n):\n    per_cell_type_count_df = train_cite_targets_df[\n        train_cite_targets_df[\"donor\"] == donor_it\n    ][[cd_it, \"day\", \"cell_type\"]]\n\n    per_cell_type_count_df = per_cell_type_count_df \\\n        .groupby([\"day\", \"cell_type\"]) \\\n        .count() \\\n        .reset_index()\n    \n    count_per_day_df = get_count_df(donor_it=donor_it, cd_it=cd_it)\n    per_cell_type_count_df = per_cell_type_count_df.join(\n        count_per_day_df, \n        \"day\", \n        how=\"left\", \n        lsuffix=\"left\"\n    )\n    per_cell_type_count_df[cd_it] = per_cell_type_count_df[f\"{cd_it}left\"] / per_cell_type_count_df[cd_it]\n    del per_cell_type_count_df[f\"{cd_it}left\"]\n    del per_cell_type_count_df[\"dayleft\"]    \n    \n    return per_cell_type_count_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:56.766690Z","iopub.execute_input":"2023-03-11T17:43:56.767401Z","iopub.status.idle":"2023-03-11T17:43:56.777085Z","shell.execute_reply.started":"2023-03-11T17:43:56.767368Z","shell.execute_reply":"2023-03-11T17:43:56.776010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@cached\ndef get_per_cell_type_mean(\n    donor_it: int = 32606,\n    cd_it: str = \"CD36\"\n):\n    per_cell_type_df = train_cite_targets_df[\n        train_cite_targets_df[\"donor\"] == donor_it\n    ][[cd_it, \"day\", \"cell_type\"]]\n\n    per_cell_type_df = per_cell_type_df \\\n        .groupby([\"day\", \"cell_type\"]) \\\n        .mean() \\\n        .reset_index()\n    \n    return per_cell_type_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:56.778573Z","iopub.execute_input":"2023-03-11T17:43:56.778919Z","iopub.status.idle":"2023-03-11T17:43:56.787424Z","shell.execute_reply.started":"2023-03-11T17:43:56.778889Z","shell.execute_reply":"2023-03-11T17:43:56.786243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@cached\ndef get_join_cd_rna(\n    donor_it: int = 32606,\n    cd_it: str = \"CD36\",\n    rna_it: str = \"ENSG00000135218_CD36\"\n):\n    cd_df = train_cite_targets_df[[cd_it]]\n    rna_df = train_cite_inputs_df[[rna_it]].join(cd_df)\n    rna_df = rna_df.join(metadata_df)\n    rna_df = rna_df[rna_df[\"donor\"] == donor_it]\n    return rna_df ","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:56.789123Z","iopub.execute_input":"2023-03-11T17:43:56.789566Z","iopub.status.idle":"2023-03-11T17:43:56.798513Z","shell.execute_reply.started":"2023-03-11T17:43:56.789506Z","shell.execute_reply":"2023-03-11T17:43:56.797529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_count_rna_cd(cd_it = \"CD36\"):\n    fig, axs = plt.subplots(4, 3)\n    fig.set_size_inches(24, 24)\n\n\n    for idx, donor_it in enumerate([32606, 13176, 31800]):\n        inner_idx = -1\n\n        inner_idx += 1\n        ax_it = axs[inner_idx][idx]\n        count_df = get_count_df(donor_it = donor_it)\n        _ = sns.barplot(\n            data=count_df,\n            x=\"day\", \n            y=cd_it,     \n            color=\"royalblue\",\n            ax=ax_it,    \n        )        \n        ax_it.set_title(f\"donor={donor_it}\")\n        ax_it.set_xlabel(f\"Day\")\n        ax_it.set_ylabel(f\"Count cells\")    \n\n\n        inner_idx += 1\n        ax_it = axs[inner_idx][idx]\n        per_cell_type_count_df = get_per_cell_type_count(\n            donor_it = donor_it\n        )\n        _ = sns.barplot(\n            data=per_cell_type_count_df,\n            x=\"day\", \n            y=cd_it, \n            hue=\"cell_type\",    \n            ax=ax_it,    \n        )\n        ax_it.set_xlabel(f\"Day\")\n        ax_it.set_ylabel(f\"The proportion of cells\")        \n\n\n        inner_idx += 1\n        ax_it = axs[inner_idx][idx]\n        mean_df = get_mean_cd_df(donor_it = donor_it)\n        _ = sns.barplot(\n            data=mean_df,\n            x=\"day\", \n            y=cd_it,  \n            color=\"royalblue\",\n            ax=ax_it,    \n        )\n        ax_it.set_xlabel(f\"Day\")\n        ax_it.set_ylabel(f\"Mean {cd_it}\")    \n\n\n        inner_idx += 1\n        ax_it = axs[inner_idx][idx]\n        per_cell_type_mean_df = get_per_cell_type_mean(\n            donor_it = donor_it\n        )\n        _ = sns.barplot(\n            data=per_cell_type_mean_df,\n            x=\"day\", \n            y=cd_it, \n            hue=\"cell_type\",    \n            ax=ax_it,    \n        )\n        ax_it.set_xlabel(f\"Day\")\n        ax_it.set_ylabel(f\"Mean {cd_it}\")      ","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:56.800113Z","iopub.execute_input":"2023-03-11T17:43:56.801317Z","iopub.status.idle":"2023-03-11T17:43:56.816115Z","shell.execute_reply.started":"2023-03-11T17:43:56.801272Z","shell.execute_reply":"2023-03-11T17:43:56.814924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_join_dist(\n    cd_it = \"CD36\",\n    rna_it = \"ENSG00000135218_CD36\",\n    is_filtered = True\n):\n    g_array = []\n    donors = [32606, 13176, 31800]\n    for idx, donor_it in enumerate(donors):\n        rna_df = get_join_cd_rna(\n            donor_it = donor_it,\n            cd_it = cd_it,\n            rna_it = rna_it\n        )\n        rna_df.sort_values(\"cell_type\", inplace=True)\n        if is_filtered:\n            rna_df = rna_df[rna_df[rna_it] > 0]\n        g = sns.jointplot(\n            data=rna_df, \n            x=cd_it, \n            y=rna_it, \n            hue=\"cell_type\"\n        )\n        g_array.append(g)\n\n\n    fig = plt.figure(figsize=(18,6))\n    gs = gridspec.GridSpec(1, len(donors))\n\n    for idx, g in enumerate(g_array):\n        _ = SeabornFig2Grid(g, fig, gs[idx])\n    #     _ = SeabornFig2Grid(g1, fig, gs[1])\n    #     _ = SeabornFig2Grid(g2, fig, gs[2])\n\n\n    gs.tight_layout(fig)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:56.817865Z","iopub.execute_input":"2023-03-11T17:43:56.819024Z","iopub.status.idle":"2023-03-11T17:43:56.828930Z","shell.execute_reply.started":"2023-03-11T17:43:56.818969Z","shell.execute_reply":"2023-03-11T17:43:56.827921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_count_rna_cd(cd_it = \"CD36\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:56.830105Z","iopub.execute_input":"2023-03-11T17:43:56.830573Z","iopub.status.idle":"2023-03-11T17:43:59.646969Z","shell.execute_reply.started":"2023-03-11T17:43:56.830536Z","shell.execute_reply":"2023-03-11T17:43:59.646075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_join_dist(\n    cd_it = \"CD36\",\n    rna_it = \"ENSG00000135218_CD36\",\n    is_filtered = True\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:43:59.648301Z","iopub.execute_input":"2023-03-11T17:43:59.648893Z","iopub.status.idle":"2023-03-11T17:44:03.873905Z","shell.execute_reply.started":"2023-03-11T17:43:59.648856Z","shell.execute_reply":"2023-03-11T17:44:03.872645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## T-SNE","metadata":{}},{"cell_type":"code","source":"target_cds = list(all_best_features_dict.keys())\nfi_array = np.array([all_best_features_dict[cd_it][\"scores\"] for cd_it in target_cds])","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:03.875455Z","iopub.execute_input":"2023-03-11T17:44:03.875830Z","iopub.status.idle":"2023-03-11T17:44:03.882922Z","shell.execute_reply.started":"2023-03-11T17:44:03.875796Z","shell.execute_reply":"2023-03-11T17:44:03.881840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.manifold import TSNE","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:03.884695Z","iopub.execute_input":"2023-03-11T17:44:03.885384Z","iopub.status.idle":"2023-03-11T17:44:04.340870Z","shell.execute_reply.started":"2023-03-11T17:44:03.885336Z","shell.execute_reply":"2023-03-11T17:44:04.339802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_embedded = TSNE(\n    n_components=2, \n    learning_rate='auto', \n    init='random', \n    perplexity=3\n).fit_transform(fi_array)\n\nplt.scatter(\n    x=fi_embedded[:, 0],\n    y=fi_embedded[:, 1],\n)\nfor i in range(len(target_cds)):\n    cd_it = target_cds[i]\n    plt.text(\n        x=fi_embedded[i, 0] + 3,\n        y=fi_embedded[i, 1],\n        s=cd_it, \n        fontdict=dict(color='red',size=10),\n        bbox=dict(facecolor='yellow',alpha=0.5)\n    )","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:04.342408Z","iopub.execute_input":"2023-03-11T17:44:04.342764Z","iopub.status.idle":"2023-03-11T17:44:04.781673Z","shell.execute_reply.started":"2023-03-11T17:44:04.342733Z","shell.execute_reply":"2023-03-11T17:44:04.780936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sim_scores_dict = defaultdict(list)\nsim_scores_dict[\"CD\"] = target_cds\n\nfor row_id, row_cd_it in enumerate(target_cds):\n    scores = []\n    row_scores = all_best_features_dict[row_cd_it][\"scores\"]\n    norm_row_scores = np.linalg.norm(row_scores)\n    for column_cd_it in target_cds:\n        column_scores = all_best_features_dict[column_cd_it][\"scores\"]\n        #sim_score = np.dot(row_scores, column_scores) / (np.linalg.norm(column_scores) * norm_row_scores)\n        sim_score = np.linalg.norm(row_scores - column_scores)\n        scores.append(sim_score)\n    \n    scores = np.array(scores)\n    #max_idxs = np.flip(np.argsort(scores))\n    max_idxs = np.argsort(scores)\n    \n    for column_id, cd_name_id in enumerate(max_idxs[1:]):\n        sim_scores_dict[f\"f{column_id}\"].append(target_cds[cd_name_id])        \n        \nsim_scores_dict_df = pd.DataFrame(sim_scores_dict)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:04.783090Z","iopub.execute_input":"2023-03-11T17:44:04.783662Z","iopub.status.idle":"2023-03-11T17:44:04.815703Z","shell.execute_reply.started":"2023-03-11T17:44:04.783628Z","shell.execute_reply":"2023-03-11T17:44:04.814012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sim_scores_dict_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:04.818128Z","iopub.execute_input":"2023-03-11T17:44:04.819151Z","iopub.status.idle":"2023-03-11T17:44:04.866829Z","shell.execute_reply.started":"2023-03-11T17:44:04.819085Z","shell.execute_reply":"2023-03-11T17:44:04.865131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CV","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\nimport catboost as cb\nimport os\nimport gc\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import defaultdict\nfrom IPython.display import FileLink        \nimport pickle as pkl\n\nimport matplotlib.gridspec as gridspec","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:04.869450Z","iopub.execute_input":"2023-03-11T17:44:04.870477Z","iopub.status.idle":"2023-03-11T17:44:04.882356Z","shell.execute_reply.started":"2023-03-11T17:44:04.870412Z","shell.execute_reply":"2023-03-11T17:44:04.880612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    \n    y2 = y_pred.copy()\n    y2 -= y2.mean(axis=0);    y2 /= y2.std(axis=0) \n    y1 = y_true.copy(); \n    y1 -= y1.mean(axis=0);    y1 /= y1.std(axis=0) \n        \n    c = (y1*y2).mean().mean()# Correlation for rescaled matrices is just matrix product and average \n        \n    return c","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:04.884881Z","iopub.execute_input":"2023-03-11T17:44:04.885881Z","iopub.status.idle":"2023-03-11T17:44:04.896701Z","shell.execute_reply.started":"2023-03-11T17:44:04.885819Z","shell.execute_reply":"2023-03-11T17:44:04.895100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_cite_inputs_df\ndel train_cite_targets_df\ndel metadata_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:04.899176Z","iopub.execute_input":"2023-03-11T17:44:04.900171Z","iopub.status.idle":"2023-03-11T17:44:05.145955Z","shell.execute_reply.started":"2023-03-11T17:44:04.900092Z","shell.execute_reply":"2023-03-11T17:44:05.144676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_input_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\ntrain_cite_target_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_targets.h5\")\nmetadata_df = pd.read_csv(\"/kaggle/input/open-problems-multimodal/metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:05.147653Z","iopub.execute_input":"2023-03-11T17:44:05.148056Z","iopub.status.idle":"2023-03-11T17:44:59.450868Z","shell.execute_reply.started":"2023-03-11T17:44:05.148018Z","shell.execute_reply":"2023-03-11T17:44:59.449643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.452278Z","iopub.execute_input":"2023-03-11T17:44:59.452586Z","iopub.status.idle":"2023-03-11T17:44:59.661321Z","shell.execute_reply.started":"2023-03-11T17:44:59.452558Z","shell.execute_reply":"2023-03-11T17:44:59.660132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# metadata_df.set_index(\"cell_id\", inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.663001Z","iopub.execute_input":"2023-03-11T17:44:59.663584Z","iopub.status.idle":"2023-03-11T17:44:59.672238Z","shell.execute_reply.started":"2023-03-11T17:44:59.663550Z","shell.execute_reply":"2023-03-11T17:44:59.671123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_target_df = train_cite_target_df.join(metadata_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.673514Z","iopub.execute_input":"2023-03-11T17:44:59.674213Z","iopub.status.idle":"2023-03-11T17:44:59.795924Z","shell.execute_reply.started":"2023-03-11T17:44:59.674180Z","shell.execute_reply":"2023-03-11T17:44:59.794963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_target_df[\"donor\"].unique()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.797467Z","iopub.execute_input":"2023-03-11T17:44:59.797922Z","iopub.status.idle":"2023-03-11T17:44:59.805791Z","shell.execute_reply.started":"2023-03-11T17:44:59.797878Z","shell.execute_reply":"2023-03-11T17:44:59.804568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd_it = \"CD36\"","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.807379Z","iopub.execute_input":"2023-03-11T17:44:59.809049Z","iopub.status.idle":"2023-03-11T17:44:59.814485Z","shell.execute_reply.started":"2023-03-11T17:44:59.809005Z","shell.execute_reply":"2023-03-11T17:44:59.813257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for day_it in [4]:\n#     for donor_it in [32606, 13176, 31800]:\n#         model_dir_path = f\"/kaggle/working/models/cv_{cd_it}_day_{day_it}_donor_{donor_it}\"\n#         model_path = model_dir_path + \"/model.cbm\"\n        \n#         if os.path.exists(model_path):\n#             continue\n        \n#         test_index = (train_cite_target_df[\"donor\"] == donor_it) & (train_cite_target_df[\"day\"] == day_it)\n\n#         model = cb.CatBoostRegressor(\n#             #task_type=\"GPU\",\n#             loss_function=\"RMSE\",\n#             metric_period=100,\n#             verbose=1,\n#         )\n        \n#         model.fit(\n#             train_cite_input_df[~test_index],\n#             train_cite_target_df[~test_index][cd_it]\n#         )      \n        \n        \n#         y_true = train_cite_target_df[test_index][cd_it].values\n\n#         y_pred = model.predict(\n#             train_cite_input_df[test_index]\n#         )\n\n#         corr_score = correlation_score(y_true, y_pred)\n#         print(f\"{cd_it}_test_{day_it}_{donor_it}, corr:\", corr_score) \n        \n        \n#         os.makedirs(model_dir_path, exist_ok=True)\n\n\n#         model.save_model(model_path)        ","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.816538Z","iopub.execute_input":"2023-03-11T17:44:59.817191Z","iopub.status.idle":"2023-03-11T17:44:59.825700Z","shell.execute_reply.started":"2023-03-11T17:44:59.817136Z","shell.execute_reply":"2023-03-11T17:44:59.824774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import time\n# while True:\n#     time.sleep(20)\n#     print(\"Sleep\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.827118Z","iopub.execute_input":"2023-03-11T17:44:59.827426Z","iopub.status.idle":"2023-03-11T17:44:59.838881Z","shell.execute_reply.started":"2023-03-11T17:44:59.827398Z","shell.execute_reply":"2023-03-11T17:44:59.837973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !ls /kaggle/working/models","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.840203Z","iopub.execute_input":"2023-03-11T17:44:59.841287Z","iopub.status.idle":"2023-03-11T17:44:59.848462Z","shell.execute_reply.started":"2023-03-11T17:44:59.841249Z","shell.execute_reply":"2023-03-11T17:44:59.847423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !zip -r /kaggle/working/models.zip /kaggle/working/models","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.850217Z","iopub.execute_input":"2023-03-11T17:44:59.850582Z","iopub.status.idle":"2023-03-11T17:44:59.858556Z","shell.execute_reply.started":"2023-03-11T17:44:59.850538Z","shell.execute_reply":"2023-03-11T17:44:59.857676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FileLink(\"models.zip\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.860114Z","iopub.execute_input":"2023-03-11T17:44:59.861203Z","iopub.status.idle":"2023-03-11T17:44:59.868317Z","shell.execute_reply.started":"2023-03-11T17:44:59.861159Z","shell.execute_reply":"2023-03-11T17:44:59.867444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CV / LGBM","metadata":{}},{"cell_type":"code","source":"catboost_fi = open_pkl(f\"{RES_BASE_DIR}/catboost_feature_importances_v5.pkl\")\ncatboost_cv_fi = open_pkl(f\"{RES_BASE_DIR}/cv_feature_importances_v5.pkl\")\nlgbm_fi = open_pkl(f\"{RES_BASE_DIR}/lgbm_feature_importances_v5.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.870080Z","iopub.execute_input":"2023-03-11T17:44:59.871232Z","iopub.status.idle":"2023-03-11T17:44:59.931766Z","shell.execute_reply.started":"2023-03-11T17:44:59.871190Z","shell.execute_reply":"2023-03-11T17:44:59.930618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"donors = list(set([it[\"donor\"] for it in catboost_cv_fi.values()]))\ndonors.sort()\n\ndays = list(set([it[\"day\"] for it in catboost_cv_fi.values()]))\ndays.sort()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.933554Z","iopub.execute_input":"2023-03-11T17:44:59.934708Z","iopub.status.idle":"2023-03-11T17:44:59.940542Z","shell.execute_reply.started":"2023-03-11T17:44:59.934659Z","shell.execute_reply":"2023-03-11T17:44:59.939307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd_it = \"CD36\"\ncv_corr_score_df = defaultdict(list)\ncv_corr_inter_df = defaultdict(list)\ncv_corr_cos_sim_df = defaultdict(list)\n\nfor day_it in days:\n    cv_corr_score_df[\"day\"].append(day_it)\n    cv_corr_inter_df[\"day\"].append(day_it)\n    cv_corr_cos_sim_df[\"day\"].append(day_it)\n    \n    for donor_it in donors:        \n        k = f\"{cd_it}_{day_it}_{donor_it}\"\n        cv_corr_score = catboost_cv_fi[k][\"corr_score\"]                \n        cv_corr_score_df[f\"donor_{donor_it}\"].append(cv_corr_score)\n        \n        assert catboost_fi[cd_it][\"columns\"] == catboost_cv_fi[k][\"columns\"]\n        \n        cv_corr_inter = intersaction_coeff(\n            catboost_fi[cd_it][\"best_100\"],\n            catboost_cv_fi[k][\"best_100\"]            \n        )\n        \n        cv_corr_inter_df[f\"donor_{donor_it}\"].append(cv_corr_inter)\n        \n        cv_corr_cos_sim = cos_sim(\n            catboost_fi[cd_it][\"scores\"],\n            catboost_cv_fi[k][\"scores\"]                        \n        )\n        cv_corr_cos_sim_df[f\"donor_{donor_it}\"].append(cv_corr_cos_sim)\n        \ncv_corr_score_df = pd.DataFrame(cv_corr_score_df)\ncv_corr_inter_df = pd.DataFrame(cv_corr_inter_df)\ncv_corr_cos_sim_df = pd.DataFrame(cv_corr_cos_sim_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.942011Z","iopub.execute_input":"2023-03-11T17:44:59.942433Z","iopub.status.idle":"2023-03-11T17:44:59.987532Z","shell.execute_reply.started":"2023-03-11T17:44:59.942392Z","shell.execute_reply":"2023-03-11T17:44:59.985801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Corr day4 test:\", catboost_fi[cd_it][\"corr_score\"])\ncv_corr_score_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:44:59.989757Z","iopub.execute_input":"2023-03-11T17:44:59.990692Z","iopub.status.idle":"2023-03-11T17:45:00.013900Z","shell.execute_reply.started":"2023-03-11T17:44:59.990632Z","shell.execute_reply":"2023-03-11T17:45:00.012329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Feature importance (N=100) intersections:\")\ncv_corr_inter_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:45:00.016495Z","iopub.execute_input":"2023-03-11T17:45:00.017561Z","iopub.status.idle":"2023-03-11T17:45:00.040328Z","shell.execute_reply.started":"2023-03-11T17:45:00.017498Z","shell.execute_reply":"2023-03-11T17:45:00.038832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Feature importance (N=100) cos sim\")\ncv_corr_cos_sim_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:45:00.042919Z","iopub.execute_input":"2023-03-11T17:45:00.044007Z","iopub.status.idle":"2023-03-11T17:45:00.068003Z","shell.execute_reply.started":"2023-03-11T17:45:00.043944Z","shell.execute_reply":"2023-03-11T17:45:00.066529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Feature importance (N=100) intersections between lgbm and catboost\")\nintersaction_coeff(\n    lgbm_fi[\"CD36\"][\"best_100\"],\n    catboost_fi[\"CD36\"][\"best_100\"]\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:45:00.070496Z","iopub.execute_input":"2023-03-11T17:45:00.071569Z","iopub.status.idle":"2023-03-11T17:45:00.091147Z","shell.execute_reply.started":"2023-03-11T17:45:00.071505Z","shell.execute_reply":"2023-03-11T17:45:00.089273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Feature importance cos sim between lgbm and catboost\")\ncos_sim(\n    lgbm_fi[\"CD36\"][\"scores\"],\n    catboost_fi[\"CD36\"][\"scores\"]\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:45:00.093595Z","iopub.execute_input":"2023-03-11T17:45:00.096214Z","iopub.status.idle":"2023-03-11T17:45:00.113298Z","shell.execute_reply.started":"2023-03-11T17:45:00.096149Z","shell.execute_reply":"2023-03-11T17:45:00.111541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd36_scores = catboost_fi[\"CD36\"][\"scores\"]\ncd36_lgbm_scores = lgbm_fi[\"CD36\"][\"scores\"]\ncd36_columns = catboost_fi[\"CD36\"][\"columns\"]\ncd36_lgbm_columns = lgbm_fi[\"CD36\"][\"columns\"]\nassert cd36_columns == cd36_lgbm_columns\n\ncd36_indexes = np.flip(np.argsort(cd36_scores))\ncd36_lgbm_indexes = np.flip(np.argsort(cd36_lgbm_scores))","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:45:00.115530Z","iopub.execute_input":"2023-03-11T17:45:00.116245Z","iopub.status.idle":"2023-03-11T17:45:00.139144Z","shell.execute_reply.started":"2023-03-11T17:45:00.116188Z","shell.execute_reply":"2023-03-11T17:45:00.137442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coeffs = []\nns = []\nfor n in tqdm(range(1, len(cd36_lgbm_indexes))):\n    catboost_features = [cd36_columns[it] for it in cd36_indexes[:n]]\n    lgbm_features = [cd36_columns[it] for it in cd36_lgbm_indexes[:n]]\n    coeff = intersaction_coeff(catboost_features, lgbm_features)\n    coeffs.append(coeff/n)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:45:00.141350Z","iopub.execute_input":"2023-03-11T17:45:00.142188Z","iopub.status.idle":"2023-03-11T17:46:54.838344Z","shell.execute_reply.started":"2023-03-11T17:45:00.142143Z","shell.execute_reply":"2023-03-11T17:46:54.836065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 6), dpi=80)\nplt.plot(coeffs)\nplt.xlabel(\"N\")\nplt.ylabel(\"Intersection / N\")\nplt.title(\"CD36: Intersection Catboost and LGBM\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:54.840988Z","iopub.execute_input":"2023-03-11T17:46:54.841365Z","iopub.status.idle":"2023-03-11T17:46:55.176479Z","shell.execute_reply.started":"2023-03-11T17:46:54.841334Z","shell.execute_reply":"2023-03-11T17:46:55.175305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:55.178182Z","iopub.execute_input":"2023-03-11T17:46:55.178638Z","iopub.status.idle":"2023-03-11T17:46:55.430483Z","shell.execute_reply.started":"2023-03-11T17:46:55.178582Z","shell.execute_reply":"2023-03-11T17:46:55.429203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catboost_lgbm_df = defaultdict(list)\nfor cd_it in catboost_fi.keys():\n    catboost_lgbm_df[\"CD\"].append(cd_it)\n    catboost_lgbm_df[\"Catboost_corr\"].append(catboost_fi[cd_it][\"corr_score\"])\n    catboost_lgbm_df[\"LGBM_corr\"].append(lgbm_fi[cd_it][\"corr_score\"])\n    \n    inter_coeff = intersaction_coeff(\n        catboost_fi[cd_it][\"best_100\"],\n        lgbm_fi[cd_it][\"best_100\"]\n    )\n    catboost_lgbm_df[\"Inter\"].append(inter_coeff)\n\n    cos_sim_coeff = cos_sim(\n        catboost_fi[cd_it][\"scores\"],\n        lgbm_fi[cd_it][\"scores\"]\n    )    \n    catboost_lgbm_df[\"CosSim\"].append(cos_sim_coeff)\n    \ncatboost_lgbm_df = pd.DataFrame(catboost_lgbm_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:55.432093Z","iopub.execute_input":"2023-03-11T17:46:55.432435Z","iopub.status.idle":"2023-03-11T17:46:55.450693Z","shell.execute_reply.started":"2023-03-11T17:46:55.432404Z","shell.execute_reply":"2023-03-11T17:46:55.449168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catboost_lgbm_df.sort_values(\"Catboost_corr\", inplace=True, ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:55.452364Z","iopub.execute_input":"2023-03-11T17:46:55.452911Z","iopub.status.idle":"2023-03-11T17:46:55.461644Z","shell.execute_reply.started":"2023-03-11T17:46:55.452859Z","shell.execute_reply":"2023-03-11T17:46:55.460265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"catboost_lgbm_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:55.463191Z","iopub.execute_input":"2023-03-11T17:46:55.463921Z","iopub.status.idle":"2023-03-11T17:46:55.494811Z","shell.execute_reply.started":"2023-03-11T17:46:55.463864Z","shell.execute_reply":"2023-03-11T17:46:55.493286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plot tree","metadata":{}},{"cell_type":"code","source":"import catboost as cb","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:55.496863Z","iopub.execute_input":"2023-03-11T17:46:55.498221Z","iopub.status.idle":"2023-03-11T17:46:55.505360Z","shell.execute_reply.started":"2023-03-11T17:46:55.498155Z","shell.execute_reply":"2023-03-11T17:46:55.503644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_path = f\"{RES_BASE_DIR}/CD36/model.cbm\"\nmodel = cb.CatBoostRegressor()\nmodel.load_model(model_path)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:55.508304Z","iopub.execute_input":"2023-03-11T17:46:55.509586Z","iopub.status.idle":"2023-03-11T17:46:55.554786Z","shell.execute_reply.started":"2023-03-11T17:46:55.509524Z","shell.execute_reply":"2023-03-11T17:46:55.553212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pool = cb.Pool(\n    train_cite_input_df, \n    train_cite_target_df[\"CD36\"],\n    feature_names=list(train_cite_input_df.columns)\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:55.557181Z","iopub.execute_input":"2023-03-11T17:46:55.558081Z","iopub.status.idle":"2023-03-11T17:46:57.867654Z","shell.execute_reply.started":"2023-03-11T17:46:55.558021Z","shell.execute_reply":"2023-03-11T17:46:57.866454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.tree_count_","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:57.869088Z","iopub.execute_input":"2023-03-11T17:46:57.869497Z","iopub.status.idle":"2023-03-11T17:46:57.875871Z","shell.execute_reply.started":"2023-03-11T17:46:57.869453Z","shell.execute_reply":"2023-03-11T17:46:57.874867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.plot_tree(tree_idx=0, pool=pool)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:57.876894Z","iopub.execute_input":"2023-03-11T17:46:57.877805Z","iopub.status.idle":"2023-03-11T17:46:59.009403Z","shell.execute_reply.started":"2023-03-11T17:46:57.877771Z","shell.execute_reply":"2023-03-11T17:46:59.008053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indexes = np.flip(np.argsort(model.get_feature_importance()))","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.011548Z","iopub.execute_input":"2023-03-11T17:46:59.012046Z","iopub.status.idle":"2023-03-11T17:46:59.057390Z","shell.execute_reply.started":"2023-03-11T17:46:59.012001Z","shell.execute_reply":"2023-03-11T17:46:59.056196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(train_cite_input_df.columns)[indexes[0]]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.058985Z","iopub.execute_input":"2023-03-11T17:46:59.059332Z","iopub.status.idle":"2023-03-11T17:46:59.073105Z","shell.execute_reply.started":"2023-03-11T17:46:59.059301Z","shell.execute_reply":"2023-03-11T17:46:59.071879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## NeurP - CV - LGBM","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nimport gc\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import defaultdict\nfrom IPython.display import FileLink        \nimport pickle as pkl\n\nimport matplotlib.gridspec as gridspec\n\nimport lightgbm as lgmb","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.075012Z","iopub.execute_input":"2023-03-11T17:46:59.075574Z","iopub.status.idle":"2023-03-11T17:46:59.082848Z","shell.execute_reply.started":"2023-03-11T17:46:59.075465Z","shell.execute_reply":"2023-03-11T17:46:59.081836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_cite_input_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\n# train_cite_target_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_targets.h5\")\n# metadata_df = pd.read_csv(\"/kaggle/input/open-problems-multimodal/metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.084481Z","iopub.execute_input":"2023-03-11T17:46:59.085369Z","iopub.status.idle":"2023-03-11T17:46:59.095996Z","shell.execute_reply.started":"2023-03-11T17:46:59.085327Z","shell.execute_reply":"2023-03-11T17:46:59.094916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# metadata_df.set_index(\"cell_id\", inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.097538Z","iopub.execute_input":"2023-03-11T17:46:59.097838Z","iopub.status.idle":"2023-03-11T17:46:59.104770Z","shell.execute_reply.started":"2023-03-11T17:46:59.097811Z","shell.execute_reply":"2023-03-11T17:46:59.103874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_cite_target_df = train_cite_target_df.join(metadata_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.105933Z","iopub.execute_input":"2023-03-11T17:46:59.106221Z","iopub.status.idle":"2023-03-11T17:46:59.114341Z","shell.execute_reply.started":"2023-03-11T17:46:59.106194Z","shell.execute_reply":"2023-03-11T17:46:59.113647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"day_it = 4\ncd_it = \"CD36\"\ncell_type = \"HSC\"\n\ntest_index = ((train_cite_target_df[\"day\"] == day_it) & (train_cite_target_df[\"cell_type\"] == cell_type))\ntrain_index = ((train_cite_target_df[\"day\"] != day_it) & (train_cite_target_df[\"cell_type\"] == cell_type))","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.115276Z","iopub.execute_input":"2023-03-11T17:46:59.115661Z","iopub.status.idle":"2023-03-11T17:46:59.131330Z","shell.execute_reply.started":"2023-03-11T17:46:59.115620Z","shell.execute_reply":"2023-03-11T17:46:59.130234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_cell_type_test_train_index(\n    day: int = 4,\n    donor: int = None,\n    is_exclude: bool = False,\n    cell_type = \"HSC\"\n):\n    test_index = None\n    train_index = None\n    if day is not None:\n        iter_index = train_cite_target_df[\"day\"] == day\n        if test_index is None:\n            test_index = iter_index            \n            train_index = ~iter_index\n        else:\n            test_index = test_index & iter_index\n            train_index = train_index & (~iter_index)\n\n    if donor is not None:\n        iter_index = train_cite_target_df[\"donor\"] == donor\n        if test_index is None:\n            test_index = iter_index\n            train_index = ~iter_index\n        else:\n            test_index = test_index & iter_index  \n            train_index = train_index & (~iter_index)\n            \n    if not is_exclude:\n        train_index = ~test_index\n            \n    if cell_type is not None:\n        iter_index = train_cite_target_df[\"cell_type\"] == cell_type\n        test_index = test_index & iter_index            \n        train_index = train_index & iter_index\n            \n    return test_index, train_index","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.134869Z","iopub.execute_input":"2023-03-11T17:46:59.135204Z","iopub.status.idle":"2023-03-11T17:46:59.143963Z","shell.execute_reply.started":"2023-03-11T17:46:59.135173Z","shell.execute_reply":"2023-03-11T17:46:59.142951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hsc_dataset_len = len(train_cite_target_df[train_cite_target_df[\"cell_type\"] == \"HSC\"])\n# cd_it = \"CD36\"\n\n# for day_it in tqdm([None, 2, 3, 4]):\n#     for donor_it in [None, 32606, 13176, 31800]:\n#         for is_exclude in [True, False]:\n#             if day_it == None and donor_it is None:\n#                 continue\n            \n#             if (day_it == None or donor_it is None) and is_exclude:\n#                 continue \n                \n#             test_index, train_index = get_cell_type_test_train_index(\n#                 day = day_it, \n#                 donor = donor_it, \n#                 is_exclude = is_exclude\n#             )\n                        \n            \n#             len_train = len(train_cite_target_df[train_index])\n#             len_test = len(train_cite_target_df[test_index])\n#             diff = hsc_dataset_len - len_train - len_test\n#             #####\n#             model = lgmb.LGBMRegressor(\n#                 verbose = 1\n#             )\n\n#             model.fit(\n#                 train_cite_input_df[train_index],\n#                 train_cite_target_df[train_index][cd_it]\n#             )\n\n#             y_true = train_cite_target_df[test_index][cd_it].values\n\n#             y_pred = model.predict(\n#                 train_cite_input_df[test_index]\n#             )    \n\n#             corr_score = correlation_score(y_true, y_pred)\n#             ######\n#             print(f\"corr={corr_score}; day={day_it} donor={donor_it} is_exclude={is_exclude} hsc_size={hsc_dataset_len} train={len_train} test={len_test} diff={diff}\")\n            \n#             model_dir_path = f\"/kaggle/working/models/\" + f\"lgmb#day={day_it}#donor={donor_it}#is_exclude={is_exclude}\"            \n#             os.makedirs(model_dir_path, exist_ok=True)\n            \n#             model_path = model_dir_path + \"/model.txt\"\n#             model.booster_.save_model(model_path)\n            \n#             del model\n#             del train_index\n#             del test_index\n#             gc.collect()\n            \n#             print(\"\")                    ","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.145742Z","iopub.execute_input":"2023-03-11T17:46:59.146147Z","iopub.status.idle":"2023-03-11T17:46:59.158141Z","shell.execute_reply.started":"2023-03-11T17:46:59.146105Z","shell.execute_reply":"2023-03-11T17:46:59.157248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## HSC VC","metadata":{}},{"cell_type":"code","source":"del train_cite_input_df\ndel train_cite_target_df\ndel metadata_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.159856Z","iopub.execute_input":"2023-03-11T17:46:59.160234Z","iopub.status.idle":"2023-03-11T17:46:59.238140Z","shell.execute_reply.started":"2023-03-11T17:46:59.160196Z","shell.execute_reply":"2023-03-11T17:46:59.236983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.239732Z","iopub.execute_input":"2023-03-11T17:46:59.240400Z","iopub.status.idle":"2023-03-11T17:46:59.496660Z","shell.execute_reply.started":"2023-03-11T17:46:59.240350Z","shell.execute_reply":"2023-03-11T17:46:59.495424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_input_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\ntrain_cite_target_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_targets.h5\")\nmetadata_df = pd.read_csv(\"/kaggle/input/open-problems-multimodal/metadata.csv\")\n\nmetadata_df.set_index(\"cell_id\", inplace=True)\ntrain_cite_target_df = train_cite_target_df.join(metadata_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:46:59.498439Z","iopub.execute_input":"2023-03-11T17:46:59.499158Z","iopub.status.idle":"2023-03-11T17:47:55.495444Z","shell.execute_reply.started":"2023-03-11T17:46:59.499111Z","shell.execute_reply":"2023-03-11T17:47:55.494419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_cell_type_test_train_index(\n    day: int = 4,\n    donor: int = None,\n    is_exclude: bool = False,\n    cell_type = \"HSC\"\n):\n    test_index = None\n    train_index = None\n    if day is not None:\n        iter_index = train_cite_target_df[\"day\"] == day\n        if test_index is None:\n            test_index = iter_index            \n            train_index = ~iter_index\n        else:\n            test_index = test_index & iter_index\n            train_index = train_index & (~iter_index)\n\n    if donor is not None:\n        iter_index = train_cite_target_df[\"donor\"] == donor\n        if test_index is None:\n            test_index = iter_index\n            train_index = ~iter_index\n        else:\n            test_index = test_index & iter_index  \n            train_index = train_index & (~iter_index)\n            \n    if not is_exclude:\n        train_index = ~test_index\n            \n    if cell_type is not None:\n        iter_index = train_cite_target_df[\"cell_type\"] == cell_type\n        test_index = test_index & iter_index            \n        train_index = train_index & iter_index\n            \n    return test_index, train_index","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:47:55.496840Z","iopub.execute_input":"2023-03-11T17:47:55.497152Z","iopub.status.idle":"2023-03-11T17:47:55.508441Z","shell.execute_reply.started":"2023-03-11T17:47:55.497123Z","shell.execute_reply":"2023-03-11T17:47:55.507678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_hsc_df = defaultdict(list)\ncd_it = \"CD36\"\nfor day_it in tqdm([None, 2, 3, 4]):\n    for donor_it in [None, 32606, 13176, 31800]:\n        for is_exclude in [True, False]:\n            if day_it == None and donor_it is None:\n                continue\n            \n            if (day_it == None or donor_it is None) and is_exclude:\n                continue\n                \n            model_path = f\"{RES_BASE_DIR}/hsc_lgmb_day={day_it}_donor={donor_it}_is_exclude={is_exclude}/model.txt\"\n            \n            if not os.path.exists(model_path):        \n                print(f\"Don't have model: {cd_it}\")\n                continue\n\n            test_index, train_index = get_cell_type_test_train_index(\n                day = day_it, \n                donor = donor_it, \n                is_exclude = is_exclude\n            )\n                \n            model = lgmb.Booster(model_file=model_path)\n            \n            y_pred = model.predict(\n                train_cite_input_df[test_index]\n            )    \n            y_true = train_cite_target_df[test_index][cd_it].values\n            \n            corr_score = correlation_score(y_true, y_pred)\n            \n            cv_hsc_df[\"day\"].append(day_it)\n            cv_hsc_df[\"donor\"].append(donor_it)\n            cv_hsc_df[\"is_exclude\"].append(is_exclude)\n            \n            cv_hsc_df[\"corr_score\"].append(corr_score)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:47:55.509783Z","iopub.execute_input":"2023-03-11T17:47:55.510099Z","iopub.status.idle":"2023-03-11T17:48:32.546878Z","shell.execute_reply.started":"2023-03-11T17:47:55.510068Z","shell.execute_reply":"2023-03-11T17:48:32.545881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_hsc_df = pd.DataFrame(cv_hsc_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:48:32.550537Z","iopub.execute_input":"2023-03-11T17:48:32.550911Z","iopub.status.idle":"2023-03-11T17:48:32.557385Z","shell.execute_reply.started":"2023-03-11T17:48:32.550877Z","shell.execute_reply":"2023-03-11T17:48:32.556224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = cv_hsc_df[cv_hsc_df[\"donor\"].isnull()]\nprint(\"Correclation day test\")\nprint(\"STD:\", np.round(df.corr_score.std(), 3), \"\\nMean: \", np.round(df.corr_score.mean(), 3))\ndf[[\"day\", \"donor\", \"corr_score\"]]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:48:32.559282Z","iopub.execute_input":"2023-03-11T17:48:32.559730Z","iopub.status.idle":"2023-03-11T17:48:32.582620Z","shell.execute_reply.started":"2023-03-11T17:48:32.559680Z","shell.execute_reply":"2023-03-11T17:48:32.581401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = cv_hsc_df[cv_hsc_df[\"day\"].isnull()]\nprint(\"Correclation donor test\")\nprint(\"STD:\", np.round(df.corr_score.std(), 3), \"\\nMean: \", np.round(df.corr_score.mean(), 3))\ndf[[\"day\", \"donor\", \"corr_score\"]]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:48:32.584062Z","iopub.execute_input":"2023-03-11T17:48:32.584410Z","iopub.status.idle":"2023-03-11T17:48:32.601099Z","shell.execute_reply.started":"2023-03-11T17:48:32.584378Z","shell.execute_reply":"2023-03-11T17:48:32.600011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = cv_hsc_df[(~cv_hsc_df[\"donor\"].isnull()) & (~cv_hsc_df[\"day\"].isnull()) & (cv_hsc_df[\"is_exclude\"] == False)]\nprint(\"Correclation day-donor is_exclude=False (full data)\")\nprint(\"STD:\", np.round(df.corr_score.std(), 3), \"\\nMean: \", np.round(df.corr_score.mean(), 3))\ndf[[\"day\", \"donor\", \"corr_score\"]]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:48:32.602676Z","iopub.execute_input":"2023-03-11T17:48:32.603220Z","iopub.status.idle":"2023-03-11T17:48:32.623893Z","shell.execute_reply.started":"2023-03-11T17:48:32.603160Z","shell.execute_reply":"2023-03-11T17:48:32.622806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = cv_hsc_df[(~cv_hsc_df[\"donor\"].isnull()) & (~cv_hsc_df[\"day\"].isnull()) & (cv_hsc_df[\"is_exclude\"] == True)]\nprint(\"Correclation day-donor is_exclude=True (part of data)\")\nprint(\"STD:\", np.round(df.corr_score.std(), 3), \"\\nMean: \", np.round(df.corr_score.mean(), 3))\ndf[[\"day\", \"donor\", \"corr_score\"]]","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:48:32.625553Z","iopub.execute_input":"2023-03-11T17:48:32.625957Z","iopub.status.idle":"2023-03-11T17:48:32.646312Z","shell.execute_reply.started":"2023-03-11T17:48:32.625923Z","shell.execute_reply":"2023-03-11T17:48:32.645572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.__version__","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:48:32.647751Z","iopub.execute_input":"2023-03-11T17:48:32.648055Z","iopub.status.idle":"2023-03-11T17:48:32.654261Z","shell.execute_reply.started":"2023-03-11T17:48:32.648028Z","shell.execute_reply":"2023-03-11T17:48:32.653135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ALL LGBM: CD->RNA","metadata":{}},{"cell_type":"code","source":"fi_df = pd.read_feather(f\"{RES_BASE_DIR}/fi_all_lgbm.feather\")","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:35.223367Z","iopub.execute_input":"2023-03-14T22:42:35.223782Z","iopub.status.idle":"2023-03-14T22:42:35.679689Z","shell.execute_reply.started":"2023-03-14T22:42:35.223750Z","shell.execute_reply":"2023-03-14T22:42:35.678658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corrs = fi_df[\"corr\"]","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:35.681784Z","iopub.execute_input":"2023-03-14T22:42:35.682948Z","iopub.status.idle":"2023-03-14T22:42:35.688869Z","shell.execute_reply.started":"2023-03-14T22:42:35.682898Z","shell.execute_reply":"2023-03-14T22:42:35.687771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6), dpi=80)\n_ = plt.hist(corrs, bins=40)\n_ = plt.title(\"All cds\")\n_ = plt.xlabel(\"corr\")\n_ = plt.ylabel(\"num\")","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:35.761156Z","iopub.execute_input":"2023-03-14T22:42:35.761841Z","iopub.status.idle":"2023-03-14T22:42:36.108340Z","shell.execute_reply.started":"2023-03-14T22:42:35.761806Z","shell.execute_reply":"2023-03-14T22:42:36.106274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cds = fi_df[\"CD\"].unique()\ndays = fi_df[\"day\"].unique()\ndonors = fi_df[\"donor\"].unique()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:36.109869Z","iopub.execute_input":"2023-03-14T22:42:36.110196Z","iopub.status.idle":"2023-03-14T22:42:36.118606Z","shell.execute_reply.started":"2023-03-14T22:42:36.110167Z","shell.execute_reply":"2023-03-14T22:42:36.117396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bast_cds = []\nfor cd_it in cds:\n    fi_cd_df = fi_df[(fi_df[\"corr\"] > 0.7) & (fi_df[\"CD\"] == cd_it)]\n    if len(fi_cd_df) == (len(days) * len(donors)):\n        bast_cds.append(cd_it)","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:36.314064Z","iopub.execute_input":"2023-03-14T22:42:36.315207Z","iopub.status.idle":"2023-03-14T22:42:36.431423Z","shell.execute_reply.started":"2023-03-14T22:42:36.315163Z","shell.execute_reply":"2023-03-14T22:42:36.430415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_fi_df = fi_df[fi_df[\"CD\"].isin(bast_cds)]\n\nbest_fi_df.groupby(\"CD\").agg({\n    \"corr\": lambda x: np.mean(x)\n})","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:36.595038Z","iopub.execute_input":"2023-03-14T22:42:36.595431Z","iopub.status.idle":"2023-03-14T22:42:36.620260Z","shell.execute_reply.started":"2023-03-14T22:42:36.595399Z","shell.execute_reply":"2023-03-14T22:42:36.619179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\ndef open_json(path):\n    with open(path, \"r\") as json_file:\n        data = json.load(json_file)\n        return data","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:36.851005Z","iopub.execute_input":"2023-03-14T22:42:36.851388Z","iopub.status.idle":"2023-03-14T22:42:36.856696Z","shell.execute_reply.started":"2023-03-14T22:42:36.851358Z","shell.execute_reply":"2023-03-14T22:42:36.855667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = open_json(f\"{RES_BASE_DIR}/msci_config_experiment.json\")\nlgbm_df = pd.read_feather(f\"{RES_BASE_DIR}/fi_all_lgbm.feather\")\nlgbm_dupl_df = pd.read_feather(f\"{RES_BASE_DIR}/fi_dupl_all_lgbm.feather\")\nlgbm_perm_df = pd.read_feather(f\"{RES_BASE_DIR}/fi_perm_all_lgbm.feather\")\n\nduplicate_columns = config[\"duplicate_columns\"]\npermutate_columns = config[\"permutate_columns\"]\n\ncolumns = config[\"columns\"]\nnorm_columns = [it.split(\"_\")[1] for it in columns]\n\nbatches_config = config[\"batches\"]","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:37.111296Z","iopub.execute_input":"2023-03-14T22:42:37.112013Z","iopub.status.idle":"2023-03-14T22:42:38.173022Z","shell.execute_reply.started":"2023-03-14T22:42:37.111976Z","shell.execute_reply":"2023-03-14T22:42:38.172010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_perm_pos(\n    fi, \n    permutate_columns = permutate_columns\n):\n    fi_sorted_index = np.argsort(fi)[::-1]\n    for idx, column in enumerate(np.array(permutate_columns)[fi_sorted_index]):\n        if column.startswith(\"PERM\"):\n            return idx    ","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:38.174696Z","iopub.execute_input":"2023-03-14T22:42:38.175020Z","iopub.status.idle":"2023-03-14T22:42:38.180973Z","shell.execute_reply.started":"2023-03-14T22:42:38.174991Z","shell.execute_reply":"2023-03-14T22:42:38.179800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def find_dupl_distance(\n    fi,\n    duplicate_columns = duplicate_columns,\n    max_pos = None\n):\n    fi_sorted_index = np.argsort(fi)[::-1]\n    sorted_columns = np.array(duplicate_columns)[fi_sorted_index]\n    column_to_idx = { column:idx for idx, column in enumerate(sorted_columns) }\n        \n    distances = []\n    for idx,column in enumerate(sorted_columns):\n        if max_pos is not None and idx > max_pos:\n            break\n        if column.startswith(\"DUPL_\"):\n            oposite_column = column[5:]\n        else:\n            oposite_column = \"DUPL_\" + column\n        \n        oposite_idx = column_to_idx[oposite_column]\n        dist = np.abs(idx - oposite_idx)\n        distances.append(dist)\n    return distances","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:38.182696Z","iopub.execute_input":"2023-03-14T22:42:38.183254Z","iopub.status.idle":"2023-03-14T22:42:38.193297Z","shell.execute_reply.started":"2023-03-14T22:42:38.183219Z","shell.execute_reply":"2023-03-14T22:42:38.192375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_sorted_columns(\n    fi,\n    columns = columns\n):\n    fi_sorted_index = np.argsort(fi)[::-1]\n    return np.array(columns)[fi_sorted_index].tolist()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:38.195180Z","iopub.execute_input":"2023-03-14T22:42:38.195772Z","iopub.status.idle":"2023-03-14T22:42:38.206901Z","shell.execute_reply.started":"2023-03-14T22:42:38.195739Z","shell.execute_reply":"2023-03-14T22:42:38.205765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perm_best_fi_df = lgbm_perm_df[lgbm_perm_df[\"CD\"].isin(bast_cds)]","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:38.208297Z","iopub.execute_input":"2023-03-14T22:42:38.208842Z","iopub.status.idle":"2023-03-14T22:42:38.219072Z","shell.execute_reply.started":"2023-03-14T22:42:38.208809Z","shell.execute_reply":"2023-03-14T22:42:38.218090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perm_pos = [find_perm_pos(fi) for fi in perm_best_fi_df.fi.tolist()]\nperm_best_fi_df[\"permutation_position\"] = perm_pos","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:38.451439Z","iopub.execute_input":"2023-03-14T22:42:38.452201Z","iopub.status.idle":"2023-03-14T22:42:39.537662Z","shell.execute_reply.started":"2023-03-14T22:42:38.452148Z","shell.execute_reply":"2023-03-14T22:42:39.536794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perm_best_fi_df","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:39.539671Z","iopub.execute_input":"2023-03-14T22:42:39.540666Z","iopub.status.idle":"2023-03-14T22:42:39.583130Z","shell.execute_reply.started":"2023-03-14T22:42:39.540602Z","shell.execute_reply":"2023-03-14T22:42:39.577147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perm_best_fi_df_agg = perm_best_fi_df.groupby(\"CD\").agg({\n    \"corr\": lambda x: np.mean(x),\n    \"permutation_position\": lambda x: np.round(np.mean(x), 2)\n})","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:39.585040Z","iopub.execute_input":"2023-03-14T22:42:39.585449Z","iopub.status.idle":"2023-03-14T22:42:39.597438Z","shell.execute_reply.started":"2023-03-14T22:42:39.585412Z","shell.execute_reply":"2023-03-14T22:42:39.596251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_fi_df_agg = best_fi_df.groupby(\"CD\").agg({\n    \"corr\": lambda x: np.mean(x)\n})","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:39.652903Z","iopub.execute_input":"2023-03-14T22:42:39.654346Z","iopub.status.idle":"2023-03-14T22:42:39.666723Z","shell.execute_reply.started":"2023-03-14T22:42:39.654292Z","shell.execute_reply":"2023-03-14T22:42:39.665416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perm_best_fi_df_agg","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:40.532490Z","iopub.execute_input":"2023-03-14T22:42:40.533568Z","iopub.status.idle":"2023-03-14T22:42:40.545376Z","shell.execute_reply.started":"2023-03-14T22:42:40.533512Z","shell.execute_reply":"2023-03-14T22:42:40.544432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_fi_df_agg[\"permutation_position\"] = perm_best_fi_df_agg[\"permutation_position\"]","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:40.993696Z","iopub.execute_input":"2023-03-14T22:42:40.994097Z","iopub.status.idle":"2023-03-14T22:42:41.000131Z","shell.execute_reply.started":"2023-03-14T22:42:40.994065Z","shell.execute_reply":"2023-03-14T22:42:40.999010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_fi_df_agg","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:41.555872Z","iopub.execute_input":"2023-03-14T22:42:41.556297Z","iopub.status.idle":"2023-03-14T22:42:41.568907Z","shell.execute_reply.started":"2023-03-14T22:42:41.556262Z","shell.execute_reply":"2023-03-14T22:42:41.567424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"perm_best_fi_df_agg","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:42.177528Z","iopub.execute_input":"2023-03-14T22:42:42.178292Z","iopub.status.idle":"2023-03-14T22:42:42.191547Z","shell.execute_reply.started":"2023-03-14T22:42:42.178254Z","shell.execute_reply":"2023-03-14T22:42:42.190394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Cross validation for best CDs","metadata":{}},{"cell_type":"code","source":"from scipy.stats import ttest_rel","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:45.510186Z","iopub.execute_input":"2023-03-14T22:42:45.510602Z","iopub.status.idle":"2023-03-14T22:42:45.515383Z","shell.execute_reply.started":"2023-03-14T22:42:45.510570Z","shell.execute_reply":"2023-03-14T22:42:45.514490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NS = [10, 20, 200, 1000, 2000, 4000, 10000, 15000]\nNAMES = [10, 20, \"perm: ~20\", 200, 1000, 2000, 4000, 10000, 15000, \"full\"]\ncv_full_df = pd.read_pickle(\"/kaggle/input/cv-full-perm-features/CV_FULL_ALL_FEATURES.pkl\")\ncv_perm_df = pd.read_pickle(\"/kaggle/input/cv-full-perm-features/CV_FULL_PERM_FEATURES.pkl\")\ncv_n_dfs = {\n    n: pd.read_pickle(f\"/kaggle/input/cv-full-perm-features/CV_FULL_PERM_FEATURES_{n}.pkl\") for n in NS\n}\n\ncv_n_dfs[\"full\"] = cv_full_df\ncv_n_dfs[\"perm: ~20\"] = cv_perm_df","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:45.822041Z","iopub.execute_input":"2023-03-14T22:42:45.822430Z","iopub.status.idle":"2023-03-14T22:42:45.904940Z","shell.execute_reply.started":"2023-03-14T22:42:45.822397Z","shell.execute_reply":"2023-03-14T22:42:45.903702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t_stats = {}\nmeans = {}\nfor name in NAMES:\n    df = cv_n_dfs[name]\n    x = df[\"corr\"].values\n    t_stat = ttest_rel(\n        cv_full_df[\"corr\"].values,\n        x        \n    )\n    t_stats[name] = t_stat\n    means[name] = np.mean(x)","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:46.069313Z","iopub.execute_input":"2023-03-14T22:42:46.069745Z","iopub.status.idle":"2023-03-14T22:42:46.085154Z","shell.execute_reply.started":"2023-03-14T22:42:46.069708Z","shell.execute_reply":"2023-03-14T22:42:46.084071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"table_df = defaultdict(list)\nfor name in NAMES:\n    corr_perc_drop = 100.0 * ((means[\"full\"] - means[name]) / means[\"full\"])\n    table_df[\"N\"].append(name)\n    table_df[\"mean\"].append(means[name])\n    table_df[\"t_stat\"].append(t_stats[name].statistic)\n    table_df[\"p_value\"].append(t_stats[name].pvalue)\n    table_df[\"corr_drop_%\"].append(corr_perc_drop)\n\ntable_df = pd.DataFrame(table_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:46.337808Z","iopub.execute_input":"2023-03-14T22:42:46.338205Z","iopub.status.idle":"2023-03-14T22:42:46.346676Z","shell.execute_reply.started":"2023-03-14T22:42:46.338172Z","shell.execute_reply":"2023-03-14T22:42:46.345371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"table_df","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:42:46.637391Z","iopub.execute_input":"2023-03-14T22:42:46.637841Z","iopub.status.idle":"2023-03-14T22:42:46.652208Z","shell.execute_reply.started":"2023-03-14T22:42:46.637803Z","shell.execute_reply":"2023-03-14T22:42:46.651083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Neurips2021, simple test/train split","metadata":{}},{"cell_type":"code","source":"neur2021_fi_df = pd.read_pickle(\"/kaggle/input/simple-neurips2021-v1/fi_neur2021_simple_test.pkl\")\nneur2021_fi_columns_df = pd.read_pickle(\"/kaggle/input/simple-neurips2021-v1/fi_neur2021_columns.pkl\")\nfi_simple_df = pd.read_pickle(\"/kaggle/input/simple-neurips2021-v1/fi_cite_seq_2022_simple_test.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:43:29.639893Z","iopub.execute_input":"2023-03-14T22:43:29.640422Z","iopub.status.idle":"2023-03-14T22:43:29.686976Z","shell.execute_reply.started":"2023-03-14T22:43:29.640376Z","shell.execute_reply":"2023-03-14T22:43:29.685808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_simple_df","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:43:36.586625Z","iopub.execute_input":"2023-03-14T22:43:36.587023Z","iopub.status.idle":"2023-03-14T22:43:36.612428Z","shell.execute_reply.started":"2023-03-14T22:43:36.586992Z","shell.execute_reply":"2023-03-14T22:43:36.610941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"neur2021_fi_df","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:43:45.286145Z","iopub.execute_input":"2023-03-14T22:43:45.286575Z","iopub.status.idle":"2023-03-14T22:43:45.312749Z","shell.execute_reply.started":"2023-03-14T22:43:45.286539Z","shell.execute_reply":"2023-03-14T22:43:45.311529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"neur_columns = neur2021_fi_columns_df.index.tolist()","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:44:23.698702Z","iopub.execute_input":"2023-03-14T22:44:23.699109Z","iopub.status.idle":"2023-03-14T22:44:23.704811Z","shell.execute_reply.started":"2023-03-14T22:44:23.699079Z","shell.execute_reply":"2023-03-14T22:44:23.703674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_cds = neur2021_fi_df[\"CD\"].tolist()\ntop_ks = [10, 100, 1000, 10000]","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:50:25.947027Z","iopub.execute_input":"2023-03-14T22:50:25.948156Z","iopub.status.idle":"2023-03-14T22:50:25.953650Z","shell.execute_reply.started":"2023-03-14T22:50:25.948098Z","shell.execute_reply":"2023-03-14T22:50:25.952392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_res_df = defaultdict(list)\nfor cd_it in best_cds:\n    ds2022_rec = fi_simple_df[fi_simple_df[\"CD\"] == cd_it].iloc[0]\n    neur2021_rec = neur2021_fi_df[neur2021_fi_df[\"CD\"] == cd_it].iloc[0]\n    \n    fi_res_df[\"CD\"].append(cd_it)\n    fi_res_df[\"corr_2022\"].append(ds2022_rec[\"corr\"])\n    fi_res_df[\"corr_neur2021\"].append(neur2021_rec[\"corr\"])\n    \n    ds2022_fi = get_sorted_columns(ds2022_rec[\"fi\"], norm_columns)\n    neur2021_fi = get_sorted_columns(neur2021_rec[\"fi\"], neur_columns)\n    \n    for top_k in top_ks:\n        inter_coeff = intersaction_coeff(\n            ds2022_fi[:top_k],\n            neur2021_fi[:top_k]\n        )\n        fi_res_df[f\"top_{top_k}\"].append(inter_coeff)\n        #break\n        \nfi_res_df = pd.DataFrame(fi_res_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:53:35.182793Z","iopub.execute_input":"2023-03-14T22:53:35.183215Z","iopub.status.idle":"2023-03-14T22:53:35.349582Z","shell.execute_reply.started":"2023-03-14T22:53:35.183181Z","shell.execute_reply":"2023-03-14T22:53:35.348444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_res_df","metadata":{"execution":{"iopub.status.busy":"2023-03-14T22:53:39.674510Z","iopub.execute_input":"2023-03-14T22:53:39.674941Z","iopub.status.idle":"2023-03-14T22:53:39.688291Z","shell.execute_reply.started":"2023-03-14T22:53:39.674905Z","shell.execute_reply":"2023-03-14T22:53:39.687462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO: Corr sim matrix for all CDs\n# TODO: FI random vect\n# TODO: https://pypi.org/project/mygene/\n\n# TODO: CV: Train Cell_type single <-> Test all\n\n# TODO: HSC <-> check CV\n\n# TODO: RNA train <-> predict RN\n# TODO: RNA train without main RNA_CD <-> predict CD\n\n# TODO: NeurP <-> split by days <-> plot only NeurP","metadata":{"execution":{"iopub.status.busy":"2023-03-11T17:48:33.338366Z","iopub.execute_input":"2023-03-11T17:48:33.339169Z","iopub.status.idle":"2023-03-11T17:48:33.344309Z","shell.execute_reply.started":"2023-03-11T17:48:33.339124Z","shell.execute_reply":"2023-03-11T17:48:33.343231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}