{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n\nimport catboost as cb\nimport os\nimport gc\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import defaultdict\nfrom IPython.display import FileLink        \nimport pickle as pkl\n\nimport matplotlib.gridspec as gridspec\nimport lightgbm as lgmb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-31T09:18:48.251365Z","iopub.execute_input":"2022-12-31T09:18:48.252315Z","iopub.status.idle":"2022-12-31T09:18:51.070200Z","shell.execute_reply.started":"2022-12-31T09:18:48.252274Z","shell.execute_reply":"2022-12-31T09:18:51.069044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    \n    y2 = y_pred.copy()\n    y2 -= y2.mean(axis=0);    y2 /= y2.std(axis=0) \n    y1 = y_true.copy(); \n    y1 -= y1.mean(axis=0);    y1 /= y1.std(axis=0) \n        \n    c = (y1*y2).mean().mean()# Correlation for rescaled matrices is just matrix product and average \n        \n    return c","metadata":{"execution":{"iopub.status.busy":"2022-12-31T09:18:51.071982Z","iopub.execute_input":"2022-12-31T09:18:51.073609Z","iopub.status.idle":"2022-12-31T09:18:51.083245Z","shell.execute_reply.started":"2022-12-31T09:18:51.073555Z","shell.execute_reply":"2022-12-31T09:18:51.082278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalization(values):\n    return (values - np.mean(values)) / np.std(values)","metadata":{"execution":{"iopub.status.busy":"2022-12-31T09:18:51.085336Z","iopub.execute_input":"2022-12-31T09:18:51.085719Z","iopub.status.idle":"2022-12-31T09:18:51.098361Z","shell.execute_reply.started":"2022-12-31T09:18:51.085652Z","shell.execute_reply":"2022-12-31T09:18:51.097200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_pkl(obj, path):\n    with open(path, \"wb\") as file:\n        pkl.dump(obj, file)","metadata":{"execution":{"iopub.status.busy":"2022-12-31T09:18:54.463098Z","iopub.execute_input":"2022-12-31T09:18:54.463560Z","iopub.status.idle":"2022-12-31T09:18:54.470412Z","shell.execute_reply.started":"2022-12-31T09:18:54.463527Z","shell.execute_reply":"2022-12-31T09:18:54.469084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_pkl(path):\n    with open(path, \"rb\") as file:\n        return pkl.load(file)    \n    \nopen_pkl = read_pkl","metadata":{"execution":{"iopub.status.busy":"2022-12-31T09:19:00.659631Z","iopub.execute_input":"2022-12-31T09:19:00.660101Z","iopub.status.idle":"2022-12-31T09:19:00.666749Z","shell.execute_reply.started":"2022-12-31T09:19:00.660067Z","shell.execute_reply":"2022-12-31T09:19:00.665411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def intersaction_coeff(list1, list2):\n    return len(set(list1).intersection(set(list2)))","metadata":{"execution":{"iopub.status.busy":"2022-12-31T09:19:07.202317Z","iopub.execute_input":"2022-12-31T09:19:07.202761Z","iopub.status.idle":"2022-12-31T09:19:07.208809Z","shell.execute_reply.started":"2022-12-31T09:19:07.202727Z","shell.execute_reply":"2022-12-31T09:19:07.207570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cos_sim(a, b):\n    cos_sim = np.dot(a, b)/(np.linalg.norm(a)*np.linalg.norm(b))\n    return cos_sim","metadata":{"execution":{"iopub.status.busy":"2022-12-31T09:19:15.359309Z","iopub.execute_input":"2022-12-31T09:19:15.359781Z","iopub.status.idle":"2022-12-31T09:19:15.366051Z","shell.execute_reply.started":"2022-12-31T09:19:15.359747Z","shell.execute_reply":"2022-12-31T09:19:15.364714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_input_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\ntrain_cite_target_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_targets.h5\")\nmetadata_df = pd.read_csv(\"/kaggle/input/open-problems-multimodal/metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-12-31T09:20:25.214582Z","iopub.execute_input":"2022-12-31T09:20:25.215499Z","iopub.status.idle":"2022-12-31T09:21:24.495965Z","shell.execute_reply.started":"2022-12-31T09:20:25.215460Z","shell.execute_reply":"2022-12-31T09:21:24.494524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df.set_index(\"cell_id\", inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-12-31T09:21:24.498822Z","iopub.execute_input":"2022-12-31T09:21:24.499240Z","iopub.status.idle":"2022-12-31T09:21:24.509332Z","shell.execute_reply.started":"2022-12-31T09:21:24.499185Z","shell.execute_reply":"2022-12-31T09:21:24.507989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_target_df = train_cite_target_df.join(metadata_df)","metadata":{"execution":{"iopub.status.busy":"2022-12-31T09:21:24.510873Z","iopub.execute_input":"2022-12-31T09:21:24.511967Z","iopub.status.idle":"2022-12-31T09:21:24.652366Z","shell.execute_reply.started":"2022-12-31T09:21:24.511931Z","shell.execute_reply":"2022-12-31T09:21:24.650918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_cell_type_test_train_index(\n    day: int = 4,\n    donor: int = None,\n    is_exclude: bool = False,\n    cell_type = \"HSC\"\n):\n    test_index = None\n    train_index = None\n    if day is not None:\n        iter_index = train_cite_target_df[\"day\"] == day\n        if test_index is None:\n            test_index = iter_index            \n            train_index = ~iter_index\n        else:\n            test_index = test_index & iter_index\n            train_index = train_index & (~iter_index)\n\n    if donor is not None:\n        iter_index = train_cite_target_df[\"donor\"] == donor\n        if test_index is None:\n            test_index = iter_index\n            train_index = ~iter_index\n        else:\n            test_index = test_index & iter_index  \n            train_index = train_index & (~iter_index)\n            \n    if not is_exclude:\n        train_index = ~test_index\n            \n    if cell_type is not None:\n        iter_index = train_cite_target_df[\"cell_type\"] == cell_type\n        test_index = test_index & iter_index            \n        train_index = train_index & iter_index\n            \n    return test_index, train_index","metadata":{"execution":{"iopub.status.busy":"2022-12-31T09:21:24.654635Z","iopub.execute_input":"2022-12-31T09:21:24.655757Z","iopub.status.idle":"2022-12-31T09:21:24.665152Z","shell.execute_reply.started":"2022-12-31T09:21:24.655720Z","shell.execute_reply":"2022-12-31T09:21:24.663592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hsc_dataset_len = len(train_cite_target_df[train_cite_target_df[\"cell_type\"] == \"HSC\"])\ncd_it = \"CD36\"\n\nfor day_it in tqdm([None, 2, 3, 4]):\n    for donor_it in [None, 32606, 13176, 31800]:\n        for is_exclude in [True, False]:\n            if day_it == None and donor_it is None:\n                continue\n            \n            if (day_it == None or donor_it is None) and is_exclude:\n                continue \n                \n            test_index, train_index = get_cell_type_test_train_index(\n                day = day_it, \n                donor = donor_it, \n                is_exclude = is_exclude\n            )\n                        \n            \n            len_train = len(train_cite_target_df[train_index])\n            len_test = len(train_cite_target_df[test_index])\n            diff = hsc_dataset_len - len_train - len_test\n            #####\n            model = lgmb.LGBMRegressor(\n                verbose = 1\n            )\n\n            model.fit(\n                train_cite_input_df[train_index],\n                train_cite_target_df[train_index][cd_it]\n            )\n\n            y_true = train_cite_target_df[test_index][cd_it].values\n\n            y_pred = model.predict(\n                train_cite_input_df[test_index]\n            )    \n\n            corr_score = correlation_score(y_true, y_pred)\n            ######\n            print(f\"corr={corr_score}; day={day_it} donor={donor_it} is_exclude={is_exclude} hsc_size={hsc_dataset_len} train={len_train} test={len_test} diff={diff}\")\n            \n            model_dir_path = f\"/kaggle/working/models/\" + f\"lgmb#day={day_it}#donor={donor_it}#is_exclude={is_exclude}\"            \n            os.makedirs(model_dir_path, exist_ok=True)\n            \n            model_path = model_dir_path + \"/model.txt\"\n            model.booster_.save_model(model_path)\n            \n            del model\n            del train_index\n            del test_index\n            gc.collect()\n            \n            print(\"\")                    ","metadata":{},"execution_count":null,"outputs":[]}]}