{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":7481061,"sourceType":"datasetVersion","datasetId":4354742}],"dockerImageVersionId":30616,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.12"},"papermill":{"default_parameters":{},"duration":2803.682528,"end_time":"2023-11-28T19:01:25.192235","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2023-11-28T18:14:41.509707","version":"2.4.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"########################################################################\n# Create required input setting-files \n########################################################################\nSETTINGS_json = {\"RAW_DATA_DIR\": \"/kaggle/input/open-problems-single-cell-perturbations/\", \n \"TRAIN_DATA_CLEAN_PATH\": \"/kaggle/input/open-problems-single-cell-perturbations/\", \"TEST_DATA_CLEAN_PATH\": \"/kaggle/input/open-problems-single-cell-perturbations/\", \n \"MODEL_CHECKPOINT_DIR\": \"./models/\", \"LOGS_DIR\": \"./logs/\", \n \"SUBMISSION_DIR\": \"./submissions/\",\n \"VERBOSE\": \"100\"}\ndisplay(SETTINGS_json )\nimport json\njson_file_path = 'SETTINGS.json'\nwith open(json_file_path, 'w') as file:\n    # Write the dictionary to the JSON file\n    json.dump(SETTINGS_json, file, indent=0)\n    \nSETTINGS_MODEL_json = {\"str_model_id\":\"CatBoost\"} #  # \"Pyboost\", 'Ridge' \n# SETTINGS_MODEL_json = {\"str_model_id\":\"Pyboost\"} # \"CatBoost\"} #  # 'Pyboost', 'Ridge' \ndisplay(SETTINGS_MODEL_json)\njson_file_path = 'SETTINGS_MODEL.json'\nwith open(json_file_path, 'w') as file:\n    # Write the dictionary to the JSON file\n    json.dump(SETTINGS_MODEL_json, file, indent=0)","metadata":{"_uuid":"7111a44d-537a-4ce1-a303-3f3d7b659d77","_cell_guid":"bf090e63-007a-49c9-9371-a938b1ff4c54","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-01-26T22:39:42.204838Z","iopub.execute_input":"2024-01-26T22:39:42.205201Z","iopub.status.idle":"2024-01-26T22:39:42.218473Z","shell.execute_reply.started":"2024-01-26T22:39:42.205171Z","shell.execute_reply":"2024-01-26T22:39:42.216946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##############################################################################\n# Imports. Get Settings. Create directories \n##############################################################################\n\nimport numpy as np\nimport pandas as pd\nimport os\nimport json\nimport time\nimport pickle \nt0start = time.time() \n\njson_file_path = 'SETTINGS.json'\n# verbose = 100\n\nwith open(json_file_path, 'r') as file:\n    dict_SETTINGS = json.load(file)\n    \nverbose =  int(dict_SETTINGS.get('VERBOSE', 0 ) )\nif verbose > 0:\n    print('Verbose = ', verbose)\n\nif verbose >=100 :\n    print('RAW_DATA_DIR:', dict_SETTINGS['RAW_DATA_DIR'])\n    print('TRAIN_DATA_CLEAN_PATH:', dict_SETTINGS['TRAIN_DATA_CLEAN_PATH'])\n    print('TEST_DATA_CLEAN_PATH:', dict_SETTINGS['TEST_DATA_CLEAN_PATH'])\n\ndirectory_path = dict_SETTINGS.get('LOGS_DIR', \"./logs/\" )\nif not os.path.exists(directory_path):\n    dict_SETTINGS['LOGS_DIR'] = directory_path\n    os.makedirs(directory_path)\n    \ndirectory_path = dict_SETTINGS.get('SUBMISSION_DIR' ,  \"./submissions/\" )\nif not os.path.exists(directory_path):\n    dict_SETTINGS['SUBMISSION_DIR'] = directory_path\n    os.makedirs(directory_path)    \n    \ndirectory_path = dict_SETTINGS.get( 'MODEL_CHECKPOINT_DIR' , \"./models/\")\nif not os.path.exists(directory_path):\n    dict_SETTINGS['MODEL_CHECKPOINT_DIR'] = directory_path\n    os.makedirs(directory_path)","metadata":{"_uuid":"83f6e477-2953-4ed4-a1e2-7e569ab59190","_cell_guid":"e79f331b-b6cc-4ca3-9686-bcfd8c618076","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-01-26T22:17:43.446834Z","iopub.execute_input":"2024-01-26T22:17:43.447125Z","iopub.status.idle":"2024-01-26T22:17:43.457752Z","shell.execute_reply.started":"2024-01-26T22:17:43.447101Z","shell.execute_reply":"2024-01-26T22:17:43.456696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##############################################################################\n# Data Load\n##############################################################################\n\nfn = os.path.join(dict_SETTINGS['RAW_DATA_DIR'], 'de_train.parquet') \ndf_de_train = pd.read_parquet(fn)# , index_col = 0)\nif verbose >= 100:\n    print(fn, df_de_train.shape)\n\nfn = os.path.join(dict_SETTINGS['RAW_DATA_DIR'], 'id_map.csv')\ndf_id_map = pd.read_csv(fn,index_col = 0)\nif verbose >=100 :\n    print(fn, df_id_map.shape )\n    \nfn = os.path.join(dict_SETTINGS['RAW_DATA_DIR'], 'sample_submission.csv')\ndf_sample_submit = pd.read_csv(fn, index_col = 0)\nif verbose >= 100:\n    print(fn, df_sample_submit.shape)\n    \n    \n#features_columns = ['cell_type','sm_name']\nselected_features_columns = ['cell_type','sm_name']  # [ 'sm_name'] # What features to consider - others not used \n    # For OP2 task we have two features - cell-type and compound - both categorical\n    # Some simple models (like Ridge) better work when one uses only one feature - 'sm_name' (=\"compound\") - so you can try that option. \n    # But seems pyboost is powerful enough to proper work with the both features.\n    # Table with results for Ridge can be found here: https://www.kaggle.com/code/alexandervc/op2-target-encoders?scriptVersionId=148576642&cellId=1\n    # Or here: https://docs.google.com/spreadsheets/d/1APN63PMaWZygVjYimK9Ivt0RvifdAU5JRYkxiDn4szw/edit?usp=sharing (sheet \"Encoders\") and discussed here:\n    \nX_submit_categorical = df_id_map[selected_features_columns] # pd.DataFrame(df_id_map, columns= features_columns )\nX_full_categorical = df_de_train[ selected_features_columns ]\n\nY_full = df_de_train.iloc[:,5:].values\nif verbose >= 100:\n    print('X_submit_categorical.shape, X_full_categorical.shape ,  Y_full.shape', X_submit_categorical.shape, X_full_categorical.shape ,  Y_full.shape)","metadata":{"_uuid":"b77e3cc4-a804-414d-a84b-1a4fdd0f7d58","_cell_guid":"19acd07a-b376-4fe9-9f2b-8f9190368f60","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-01-26T22:17:43.458945Z","iopub.execute_input":"2024-01-26T22:17:43.459236Z","iopub.status.idle":"2024-01-26T22:17:47.161464Z","shell.execute_reply.started":"2024-01-26T22:17:43.459211Z","shell.execute_reply":"2024-01-26T22:17:47.160359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##############################################################################\n# CV-schemes\n##############################################################################\n\nimport random \nfrom sklearn.model_selection import KFold\n\n# Prepare indexes for AmbrosM scheme:\n# For AmbrosM scheme: \nfolds_index_data_AmbrosM = [ ]\ntrain_sm_names = ['Idelalisib', 'Crizotinib', 'Linagliptin', 'Palbociclib', 'Dabrafenib', 'Alvocidib', 'LDN 193189', 'R428', 'Porcn Inhibitor III', \n  'Belinostat', 'Foretinib', 'MLN 2238', 'Penfluridol', 'Dactolisib', 'O-Demethylated Adapalene', 'Oprozomib (ONX 0912)', 'CHIR-99021']\nlist_fold_ids =  ['NK cells', 'T cells CD4+', 'T cells CD8+', 'T regulatory cells'] \nfor fold_id in list_fold_ids:\n    mask_va = (df_de_train.cell_type == fold_id) & ~df_de_train.sm_name.isin(train_sm_names)\n    mask_tr = ~mask_va # 485 or 487 training rows\n    IX_train = np.where( mask_tr > 0 )[0]\n    IX_test = np.where( mask_va > 0 )[0]\n    #print(fold_id,  len(IX_test), type(IX_test), IX_test[:3], len(IX_train), type(IX_train), IX_train[:3]  )\n    folds_index_data_AmbrosM.append( [IX_train, IX_test ])\n\n# For MT CV scheme:\nfolds_index_data_MT = []\nfold_to_compounds = {0: ['Alvocidib', 'Belinostat', 'Foretinib', 'LDN 193189',  'Linagliptin', 'O-Demethylated Adapalene'],\n 1: ['Dabrafenib', 'Dactolisib', 'Idelalisib', 'MLN 2238', 'Palbociclib', 'Porcn Inhibitor III'],\n 2: ['CHIR-99021', 'Crizotinib', 'Oprozomib (ONX 0912)', 'Penfluridol',  'R428']}\nfor fold_id in [0,1,2]:\n    mask_va = df_de_train['cell_type'].isin(['Myeloid cells', 'B cells']) & df_de_train['sm_name'].isin(fold_to_compounds[fold_id])\n    mask_tr = ~mask_va    \n    IX_train = np.where( mask_tr > 0 )[0]\n    IX_test = np.where( mask_va > 0 )[0]\n    #print(fold_id,  len(IX_test), type(IX_test), IX_test[:3], len(IX_train), type(IX_train), IX_train[:3]  )\n    folds_index_data_MT.append( [IX_train, IX_test ])    \n    \n    \ndict_folds_Antonina = {0: [4, 15, 18, 41, 47, 53, 56, 57, 59, 61, 62, 66, 78, 83, 86, 93, 100, 110, 115, 116, 117, 120, 123, 131, 148, 152, 161, 185, 189, 193, 197, 198, 205, 207, 209, 210, 213, 218, 219, 225, 228, 230, 250, 252, 257, 263, 286, 292, 294, 304, 306, 310, 325, 327, 336, 338, 339, 350, 351, 356, 357, 360, 362, 369, 379, 385, 395, 419, 424, 430, 432, 434, 438, 441, 443, 445, 448, 452, 456, 465, 467, 471, 472, 474, 498, 508, 523, 534, 542, 544, 574, 581, 584, 589, 601, 607], 1: [1, 3, 5, 14, 24, 38, 40, 49, 50, 54, 55, 60, 79, 82, 87, 102, 112, 114, 118, 119, 127, 132, 146, 149, 166, 179, 183, 184, 186, 190, 192, 201, 203, 204, 216, 227, 242, 253, 261, 264, 271, 289, 295, 296, 299, 300, 309, 323, 324, 329, 333, 337, 354, 387, 390, 393, 406, 408, 415, 416, 420, 421, 422, 436, 442, 449, 453, 458, 462, 463, 469, 484, 487, 489, 490, 492, 497, 501, 505, 506, 509, 513, 515, 516, 546, 548, 566, 569, 585, 587, 588, 592, 593, 595, 600, 605], 2: [0, 16, 19, 44, 48, 51, 52, 58, 63, 65, 70, 81, 85, 88, 89, 92, 111, 113, 122, 125, 135, 137, 139, 150, 164, 175, 191, 202, 206, 211, 217, 220, 221, 229, 239, 243, 247, 248, 251, 258, 259, 260, 267, 270, 274, 281, 288, 290, 291, 303, 322, 326, 328, 332, 347, 359, 364, 370, 383, 384, 389, 392, 401, 423, 428, 454, 455, 461, 464, 494, 496, 504, 510, 511, 524, 525, 526, 530, 540, 541, 551, 565, 568, 570, 572, 573, 575, 577, 579, 580, 597, 598, 603, 606, 610, 611, 613], 3: [21, 22, 26, 36, 42, 45, 46, 64, 67, 69, 71, 80, 84, 90, 101, 103, 134, 136, 140, 147, 151, 160, 163, 165, 167, 174, 176, 177, 180, 181, 182, 187, 188, 195, 196, 199, 200, 212, 215, 226, 231, 240, 241, 249, 255, 256, 262, 265, 269, 285, 287, 297, 298, 307, 312, 319, 321, 330, 340, 342, 344, 345, 348, 358, 366, 368, 382, 391, 400, 403, 404, 405, 425, 427, 431, 433, 435, 444, 450, 451, 457, 459, 460, 473, 485, 499, 500, 502, 503, 507, 512, 514, 527, 528, 529, 531, 543, 547, 553, 554, 564, 571, 576, 583, 586, 591, 596, 602, 604, 608, 609], 4: [2, 6, 7, 17, 20, 23, 25, 27, 28, 29, 37, 39, 43, 68, 91, 121, 124, 126, 128, 129, 130, 133, 138, 153, 162, 178, 194, 208, 214, 222, 223, 224, 232, 244, 245, 246, 254, 266, 268, 272, 273, 282, 283, 284, 293, 301, 302, 305, 308, 311, 320, 331, 334, 335, 341, 343, 346, 349, 352, 353, 355, 361, 363, 365, 367, 377, 378, 380, 381, 386, 388, 394, 396, 397, 398, 399, 402, 407, 417, 418, 426, 429, 437, 439, 440, 446, 447, 466, 468, 470, 481, 482, 483, 486, 488, 491, 493, 495, 532, 533, 545, 549, 550, 552, 555, 562, 563, 567, 578, 582, 590, 594, 599, 612], 5: [8, 9, 10, 11, 12, 13, 30, 31, 32, 33, 34, 35, 72, 73, 74, 75, 76, 77, 94, 95, 96, 97, 98, 99, 104, 105, 106, 107, 108, 109, 141, 142, 143, 144, 145, 154, 155, 156, 157, 158, 159, 168, 169, 170, 171, 172, 173, 233, 234, 235, 236, 237, 238, 275, 276, 277, 278, 279, 280, 313, 314, 315, 316, 317, 318, 371, 372, 373, 374, 375, 376, 409, 410, 411, 412, 413, 414, 475, 476, 477, 478, 479, 480, 517, 518, 519, 520, 521, 522, 535, 536, 537, 538, 539, 556, 557, 558, 559, 560, 561]}\n\nfolds_index_data_Antonina = []    \nfor i in range(5):  \n    l_valid = np.array( dict_folds_Antonina[i] )\n    l_train = np.array( [k for k in range(614) if k not in l_valid ] )\n    folds_index_data_Antonina.append( [l_train, l_valid ])    \nprint( len(folds_index_data_Antonina ) )\n\nclass KFold_custom:\n    '''\n    Class with similar to sklearn \"KFold\" class, interface.\n    Support of CV schemes proposed by AmbrosM, MT, Kishan and standard sklearn KFold\n    Supports options to return IX_train,IX_valid,IX_test - triplet - if valid_size > 0\n    Examples:\n    kf = KFold_custom('AmbrosM'):     \n    for i_fold,(IX_train,IX_test)   in enumerate( kf.split()) :\n        pass\n    kf = KFold_custom('AmbrosM', valid_size = 0.1 )     \n    for i_fold,(IX_train ,IX_valid,IX_test)   in enumerate( kf.split()) :\n        pass\n        \n    kf = KFold_custom(CV_scheme = 'Random') # wrapper for sklearn Kfolds with shuffle = True\n    kf = KFold_custom(CV_scheme = CV_scheme, random_state = 42 ) # fixing random_state ensures reprodicibility of folds \n    '''\n    def __init__(self, CV_scheme,  valid_size = 0, n_splits=5, random_state = 42, verbose = 0 ):\n        self.CV_scheme = CV_scheme\n        self.CV_scheme_inf =  CV_scheme\n        self.valid_size = np.clip(valid_size,0,1)\n        self.random_state = random_state\n\n        self.folds_index_data = [ [np.arange(0), np.arange(0), np.arange(0)] ] # Example data\n        self.n_splits = 1 # Example data\n        if CV_scheme == 'Kishan1':\n            # One fold scheme which contains train, valid, test parts \n            IX_train_Kishan1 = np.arange(429)\n            IX_val_Kishan1 = np.arange(429,521)\n            IX_test_Kishan1 = np.arange(521,614)\n            self.n_splits = 1\n            self.folds_index_data =  [   [IX_train_Kishan1, IX_val_Kishan1, IX_test_Kishan1]  ]\n        elif CV_scheme == 'Tonya':\n            self.n_splits = 5\n            self.folds_index_data = folds_index_data_Antonina\n        elif CV_scheme == 'AmbrosM':\n            self.n_splits = 4\n            self.folds_index_data = folds_index_data_AmbrosM\n        elif CV_scheme == 'MT':\n            self.n_splits = 3\n            self.folds_index_data = folds_index_data_MT\n        elif 'Random'.lower() in CV_scheme.lower():\n            self.n_splits = n_splits\n            self.CV_scheme_inf = 'Random_'+str(n_splits) +'_'+str(random_state )\n            kf = KFold(n_splits=n_splits, random_state = random_state, shuffle=True )#,\n            self.folds_index_data = list( kf.split( np.arange(614) ) )\n        elif 'Full'.lower() in CV_scheme.lower():\n            # Just return the full set as both train and test - it useful to re-train the model on the entire data\n            IX_train_full = np.arange(614)\n            IX_test_full = np.arange(614)\n            self.n_splits = 1\n            self.folds_index_data = [   [IX_train_full, IX_test_full]  ]\n        else:\n            s = 'Uncrecognized CV_scheme ' + str(CV_scheme) \n            raise ValueError(s)\n\n        self.verbose = verbose \n        if verbose >= 10:\n            print(self.CV_scheme, self.n_splits, len(self.folds_index_data) )\n            #print(self.folds_index_data)\n            \n    def split(self, X=None):\n        '''\n        X - NOT used, just for compatibility with sklearn \n        '''\n        for item in self.folds_index_data:\n            if self.CV_scheme in ['Kishan1']:\n                if self.valid_size > 0:\n                    yield item[0],item[1],item[2]\n                else :\n                    yield np.array( list(set(item[0])|set(item[1]))  ) , item[2] # Return train, test only \n            elif self.CV_scheme in ['Tonya', 'AmbrosM','MT', 'Random', 'Full']:\n                if self.valid_size == 0:\n                    yield item[0],item[1]\n                else:\n                    # Split \"full-train\"->( real-train, valid ) \n                    index_for_real_train_part = int(  (1-self.valid_size) * len(item[0]) )\n                    if self.random_state is None:\n                        IX_train =  item[0][:index_for_real_train_part]\n                        IX_valid =  item[0][index_for_real_train_part:]\n                    elif self.random_state == -1:\n                        p = np.random.permutation(len(item[0])) \n                        IX_train =  item[0][p][:index_for_real_train_part]\n                        IX_valid =  item[0][p][index_for_real_train_part:]\n                    else:\n                        np.random.seed(self.random_state) # Set temporary random seed\n                        p = np.random.permutation(len(item[0])) \n                        np.random.seed(random.randint(0,30000)) # Randomize seed again - use Python random, not numpy       \n                        IX_train =  item[0][p][:index_for_real_train_part]\n                        IX_valid =  item[0][p][index_for_real_train_part:]\n                    yield IX_train, IX_valid, item[1]\n                    \n        \n    def get_list_main_CV_schemes(self):\n        return ['AmbrosM','MT','Kishan1','Random']\n    def get_n_splits(self, X=np.array([])):\n        return self.n_splits","metadata":{"_uuid":"5802e130-a00e-4792-a573-231a2e7634d9","_cell_guid":"c0064c4b-2097-4e09-bbe7-b34ecb10e4d0","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-01-26T22:17:47.164154Z","iopub.execute_input":"2024-01-26T22:17:47.164527Z","iopub.status.idle":"2024-01-26T22:17:47.214362Z","shell.execute_reply.started":"2024-01-26T22:17:47.164496Z","shell.execute_reply":"2024-01-26T22:17:47.213253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"json_file_path_for_model_settings =  'SETTINGS_MODEL.json'\nif os.path.exists(json_file_path_for_model_settings):\n    with open(json_file_path_for_model_settings, 'r') as file:\n        dict_SETTINGS_MODEL = json.load(file)    \n    str_model_id = dict_SETTINGS_MODEL.get( \"str_model_id\" , \"CatBoost\" )\nelse:\n    str_model_id = 'CatBoost'    \nif verbose >= 1:\n    print( str_model_id )\n\n########################################################################\n# Modeling params Pyboost\n########################################################################\n\nif str_model_id =='Pyboost':\n    CV_scheme = 'Random' #AmbrosM\n    max_depth = 10  #  \n    ntrees = 5000 # \n    subsample = 1 # From 0 to 1\n    colsample = 0.35  # From 0 to 1 \n    lr = 0.01  # From 0 to 1 \n    n_components = 50 \n    \n    import subprocess\n    command = \"pip install py-boost\"\n    result = subprocess.run(command, shell=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True)\n    if result.returncode == 0:\n        if verbose >= 1:\n            print(\"Command  pip install py-boost executed successfully.\")\n            print(\"Output:\")\n            print(result.stdout)\n    else:\n        print(\"Error executing the command pip install py-boost.\")\n        print(\"Error message:\")\n        print(result.stderr)\n    \n    from py_boost import GradientBoosting, TLPredictor, TLCompiledPredictor\n    from py_boost.cv import CrossValidation\n    model = GradientBoosting('mse', ntrees=ntrees, lr=lr,  max_depth=max_depth  , subsample=subsample, colsample=colsample, min_data_in_leaf=1, min_gain_to_split=0, verbose=100  )           \n\n########################################################################\n# Modeling params Pyboost\n########################################################################\n\nif str_model_id =='CatBoost':\n    import catboost\n    from catboost import CatBoostRegressor, Pool\n    from sklearn.multioutput import MultiOutputRegressor\n\n    CV_scheme =  'Tonya'# 'Random' #AmbrosM\n    max_depth = 6  #  \n    ntrees = 250 # \n    subsample = 1 # From 0 to 1\n    colsample = 0.5  # From 0 to 1 \n    lr = 0.03  # From 0 to 1 \n    n_components = 30 \n    # tsvd30_modelCATB_NI250_MD6_LR0.03_SS1_CS0.5_encQuantileEncoder_quantile0.8_Dr_CT_SB1_Tonya # CT - cell-type, Dr - Drug - both included\n    # 'min_data_in_leaf': 5, 'loss_function': 'RMSE', 'random_strength': None, 'random_seed': 42, 'l2_leaf_reg': 1, \n\n    params_model = { 'iterations': ntrees,  'depth': max_depth , 'learning_rate' : lr,'subsample':subsample, 'colsample_bylevel': colsample,\n                   'min_data_in_leaf': 5, 'loss_function': 'RMSE', 'random_strength': None, 'random_seed': 42, 'l2_leaf_reg': 1 }\n    model = MultiOutputRegressor( CatBoostRegressor( verbose = 0, **params_model ) )","metadata":{"_uuid":"6aebcd07-6c02-458b-94c8-cce4575a5b68","_cell_guid":"eb145ba9-279d-4285-a382-2bdd76a62135","collapsed":false,"execution":{"iopub.status.busy":"2024-01-26T22:17:47.215787Z","iopub.execute_input":"2024-01-26T22:17:47.216696Z","iopub.status.idle":"2024-01-26T22:17:47.232681Z","shell.execute_reply.started":"2024-01-26T22:17:47.216646Z","shell.execute_reply":"2024-01-26T22:17:47.230835Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##############################################################################\n# Modeling\n##############################################################################\nfrom sklearn.linear_model import Ridge\nfrom sklearn.decomposition import TruncatedSVD\nfrom sklearn.metrics import r2_score\nfrom sklearn.metrics import mean_absolute_error\nimport category_encoders as ce\n\nflag_save_each_model_submit = True\n\nif verbose >= 10:\n    print('CV_scheme:', CV_scheme); print(); \n# Initialize CV-scheme generator:\nvalid_size = 0 \nkf = KFold_custom(CV_scheme, valid_size = valid_size)   \n\n\ndf_stat = pd.DataFrame(); # report with scores will be here \ni_total = 0\n\n\n#Initialization of Arrays for OOF and Test Predictions\noof_preds = np.zeros(Y_full.shape)\nall_Y_test_preds = np.zeros((255, Y_full.shape[1], kf.get_n_splits()))\n\n\nreducer = TruncatedSVD(n_components=n_components, n_iter=7, random_state=42)\n\n# ------------------------------------------  Encoder   ------------------------------------------\n# One can tune params for encoders also. In current version we will NOT do it.\nsmoothing = 1 # [0,1,5,8,9,10,]#11,12,15, 20,50,100,200,500, 1000,10_000, np.inf] # For Target Encoder\n\nenc = ce.QuantileEncoder(quantile =.8)\n# if verbose >= 10:\n#     print('smoothing:', smoothing, )\n#enc = ce.LeaveOneOutEncoder(  )\n\ni_total += 1\n\nY_submit_pred = np.zeros( (255,  Y_full.shape[1])   ); i_blend_for_submit = 0 \n\n\n# Fold-wise metrics will be stored here:\nlist_mrrmse = [];list_r2 = []; list_mae = []  # Fold-wise metrics\n\nt0_single_model_all_folds = time.time() # Timer starts. All folds computation timing.\n\n#  ------------------------------------------ Loop over folds  ------------------------------------------\nfor i_fold,(IX_train,IX_test)   in enumerate( kf.split()) :\n    t0_one_fold = time.time()\n\n    if str_model_id != 'CatBoost':\n        m_tmp1 = df_de_train['cell_type'] == 'T cells CD8+'\n        l_tmp1 = list(np.where(m_tmp1>0)[0])\n        IX_train = [i for i in IX_train if i not in l_tmp1]\n    else:\n        #np.random.shuffle(IX_train) #  np.array( [i for i in IX_train if df_de_train['cell_type'].iat[i] != 'T cells CD8+'  ] )\n        #IX_train = IX_train[:-10]\n        # Config 8: in V64 \n        IX_train = np.arange(614 )\n        IX_train = np.array( [i for i in IX_train if df_de_train['cell_type'].iat[i] != 'T cells CD8+'  ] )\n        np.random.shuffle(IX_train)\n        IX_train = IX_train[:-3]        \n        \n\n    # ------------------- Get train, test. (Also technical: coversion to numpy if necessary) -------------------\n    if hasattr(X_full_categorical,'values'):\n        X_train_categorical,X_test_categorical = X_full_categorical.values[IX_train], X_full_categorical.values[IX_test]\n    else:\n        X_train_categorical,X_test_categorical = X_full_categorical[IX_train], X_full_categorical[IX_test]\n    if hasattr(X_submit_categorical,'values'):\n        X_submit_categorical_loc = X_submit_categorical.values\n    else:\n        X_submit_categorical_loc = X_submit_categorical\n\n\n    #  Dimensional reduction for targets  \n    Y_train_red = reducer.fit_transform(Y_full[IX_train] ) \n\n\n    # ------------------------------------------ Initialize the model  ------------------------------------------\n    # Pay attention - it is important to make initialization of py-boost for each fold\n\n    # Use Ridge for simple/fast tests and debuging:\n    if  str_model_id == 'Ridge':\n        model = Ridge(); #str_model_id = 'Ridge'\n    elif  str_model_id == 'Pyboost':\n        # str_model_id = 'Pyboost'; \n        model = GradientBoosting('mse', ntrees=ntrees, lr=lr,  max_depth=max_depth  , subsample=subsample, colsample=colsample, min_data_in_leaf=1, min_gain_to_split=0, verbose=100  )           \n    elif  str_model_id == 'CatBoost':\n        model = MultiOutputRegressor( CatBoostRegressor( verbose = 0, **params_model ) )  \n\n    if verbose >= 100:\n        print('i_total, n_components, ntrees, lr, max_depth, colsample, subsample', i_total, n_components,  ntrees, lr, max_depth, colsample, subsample  )\n\n\n\n    ### -------------- Target Encoding -------------------------------------------------------------------\n    # Encoding is done by each of the reduced targets \n    # One can add other features here\n    # category encoders package can accept only 1-dimensional \"y\", so we need a loop and  encode using Y[:,i] one by one for all \"i\" \n    for i_target in range(Y_train_red.shape[1]):\n        if i_target == 0:\n            X_train_encoded = enc.fit_transform(X_train_categorical, Y_train_red[:,i_target])\n            X_test_encoded = enc.transform(X_test_categorical)\n            X_submit_encoded = enc.transform(X_submit_categorical_loc)\n        else:\n            X_encoded_tmp = enc.fit_transform(X_train_categorical, Y_train_red[:,i_target])\n            X_train_encoded = np.concatenate( [X_train_encoded, X_encoded_tmp], axis = 1)\n            X_encoded_tmp = enc.transform(X_test_categorical)\n            X_test_encoded = np.concatenate( [X_test_encoded, X_encoded_tmp], axis = 1)\n            X_encoded_tmp = enc.transform(X_submit_categorical_loc)\n            X_submit_encoded = np.concatenate( [X_submit_encoded, X_encoded_tmp], axis = 1)\n\n    if verbose >= 10:\n        print('fold:', i_fold, 'X_train_encoded.shape,X_test_encoded.shape,  X_train_categorical.shape, X_test_categorical.shape, Y_train_red.shape', X_train_encoded.shape,X_test_encoded.shape, X_train_categorical.shape, X_test_categorical.shape, Y_train_red.shape)\n\n\n    ### ---------------------------------- Train the model -----------------------------------------------------------------------\n    model.fit(X_train_encoded, Y_train_red)\n\n    ### ---------------------------------- Prediction for validation  -------------------------------------------------------------\n    Y_test_pred_red = model.predict( X_test_encoded )\n    ### ---------------------------------- Inverse dimensional reduction  ---------------------------------------------------------\n    Y_test_pred = reducer.inverse_transform( Y_test_pred_red )\n\n    ### ---------------------------------- Prediction, Inverse-reducer  for Submit part---------------------------------------------\n    Y_submit_pred_from_one_fold =  reducer.inverse_transform( model.predict( X_submit_encoded ) )\n    ### ---------------------------------- Average submit part from all folds          ---------------------------------------------\n    Y_submit_pred =  (Y_submit_pred * i_blend_for_submit + Y_submit_pred_from_one_fold )/(i_blend_for_submit + 1)\n    i_blend_for_submit += 1\n\n    ### ----------------------------------      Scoring computation for local test     ---------------------------------------------\n    Y_test = Y_full[IX_test]\n    mrrmse = np.sqrt(np.square(Y_test - Y_test_pred).mean(axis=1)).mean();  r2 = r2_score( Y_test , Y_test_pred ); mae_1_fold = mean_absolute_error( Y_test , Y_test_pred  )\n    if verbose >= 10:\n        print(i_total, 'fold:', i_fold, 'mrrmse:', np.round(mrrmse,4), 'r2:', np.round(r2,4),  'mae:', np.round(mae_1_fold,4),  )\n    list_mrrmse.append(mrrmse); list_r2.append(r2); list_mae.append(mae_1_fold)\n\n\n    ### Loop Through Hyperparameters and Folds\n    Y_test_pred_red = model.predict(X_test_encoded)\n    Y_test_pred = reducer.inverse_transform(Y_test_pred_red)\n    Y_submit_pred_from_one_fold = reducer.inverse_transform(model.predict(X_submit_encoded))\n\n    oof_preds[IX_test] = Y_test_pred\n    all_Y_test_preds[:, :, i_fold] = Y_submit_pred_from_one_fold\n    \n    \n    ### ----------------------------------      Save model data - checkpoint     ---------------------------------------------\n    \n    directory_path = dict_SETTINGS['MODEL_CHECKPOINT_DIR'] + 'fold'+str(i_fold) +'_'+ str_model_id  +'/' \n    if not os.path.exists(directory_path):\n        os.makedirs(directory_path)\n    file_path = os.path.join(directory_path, 'model.pkl')\n    with open(file_path, 'wb') as file:\n        pickle.dump(model, file)\n    file_path = os.path.join(directory_path, 'reducer.pkl')\n    with open(file_path, 'wb') as file:\n        pickle.dump(reducer, file)\n    file_path = os.path.join(directory_path, 'encoder.pkl')\n    with open(file_path, 'wb') as file:\n        pickle.dump(enc, file)\n        \n    \n    \n    # ------------------------------------ END of fold-loop -----------------------------------------------------------------------\n\n\ndef save_submit(Y_submit_pred, str_fn_info,  dict_params = {} ,    verbose = 0):\n    '''\n    Save submit to csv - optionally include into filename the params of the model.\n    First column should be named \"id\" - do not forget.\n    Data - 255x18211: samples x genes\n    '''\n\n    df_submit = pd.DataFrame(Y_submit_pred, columns = df_de_train.columns[5:])\n    df_submit.index.name = 'id'\n    if verbose >= 10:\n        print('df_submit.shape:', df_submit.shape, 'df_submit: ' )\n        #display(df_submit.head(2))\n    str_detailed_info = ''\n    for k in dict_params:\n        val = dict_params[k]\n        str_detailed_info += '_'+str(k)+str(val).replace('.','')\n    fn_save = 'submission_' + str_fn_info + str_detailed_info + '.csv' \n    fn_save = os.path.join(dict_SETTINGS['SUBMISSION_DIR'], fn_save ) \n    if verbose >= 1:\n        print('Submit saved to:', fn_save)\n    df_submit.to_csv(fn_save)   \n    \n\n# ------------------------------------------------ Optional. Save Submission to CSV ----------------------------------------    \nif flag_save_each_model_submit :\n    dict_params_loc = {'max_depth':max_depth, 'ntrees':ntrees, 'lr':lr, 'subsample':subsample, 'colsample':colsample, \n                       'n_components':n_components  }\n    save_submit(Y_submit_pred, str_fn_info = str_model_id,  dict_params = dict_params_loc,     verbose = verbose   )\n\n\n# ------------------------------------------------ Optional. Save Scores, other statistics ----------------------------------------    \n# Statistics/reporting part    \n# Save scores,params to the report datataframe    \nif verbose >= 1:\n    print(i_total, 'Average mrrmse:', np.round(np.mean(list_mrrmse ),4 ),  'Average r2:', np.round(np.mean(list_r2 ),4 ),  'Average mae:', np.round(np.mean(list_r2 ),4 ))\nIX = len(df_stat)+1\ndf_stat.loc[IX,'mrrmse'] =  np.mean(list_mrrmse )\ndf_stat.loc[IX,'r2'] =  np.mean(list_r2 ) \ndf_stat.loc[IX,'mae'] =  np.mean(list_mae ) \ndf_stat.loc[IX,'model'] =  str_model_id\ndf_stat.loc[IX,'CV_scheme'] =  str(CV_scheme)\ndf_stat.loc[IX,'n_components'] = n_components\ndf_stat.loc[IX,'max_depth'] =  max_depth\ndf_stat.loc[IX,'ntrees'] = ntrees\ndf_stat.loc[IX,'lr'] = lr\ndf_stat.loc[IX,'colsample'] =  colsample\ndf_stat.loc[IX,'subsample'] =   subsample\nfor i_tmp in range(len(list_mrrmse)):\n    df_stat.loc[IX,'mrrmse '+CV_scheme + ' ' + str(i_tmp)] =  list_mrrmse[i_tmp]\nfor i_tmp in range(len(list_r2)):\n    df_stat.loc[IX,'r2 '+CV_scheme + ' ' + str(i_tmp)] =  list_r2[i_tmp]\nfor i_tmp in range(len(list_mae)):\n    df_stat.loc[IX,'mae '+CV_scheme + ' ' + str(i_tmp)] =  list_mae[i_tmp]\ndf_stat.loc[IX,'Time'] = np.round(time.time() - t0_single_model_all_folds,1)  \n\nif verbose >= 1:\n    print(df_stat ) \n    \nfn = os.path.join(dict_SETTINGS['LOGS_DIR'], 'df_stat'+'_'+ str_model_id +'.csv') \ndf_stat.sort_values(df_stat.columns[0] ).to_csv(fn)","metadata":{"_uuid":"6046cfbc-bfe2-4400-90f0-094d06131e35","_cell_guid":"9ea89a63-3a15-42a1-bac9-a930491edba1","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-01-26T22:17:47.234528Z","iopub.execute_input":"2024-01-26T22:17:47.234853Z","iopub.status.idle":"2024-01-26T22:19:00.289597Z","shell.execute_reply.started":"2024-01-26T22:17:47.234817Z","shell.execute_reply":"2024-01-26T22:19:00.288507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if verbose >= 1:\n    print('%.1f seconds passed total '%(time.time()-t0start) )\n    print('%.1f minutes passed total '%( (time.time()-t0start)/60)  )\n    print('%.2f hours passed total '%( (time.time()-t0start)/3600)  )","metadata":{"_uuid":"1ac0b1bc-4424-47fc-a488-8d14fee4cad2","_cell_guid":"466fa058-e41f-45f5-9329-1f566766066e","collapsed":false,"papermill":{"duration":0.039821,"end_time":"2023-11-28T19:01:21.666987","exception":false,"start_time":"2023-11-28T19:01:21.627166","status":"completed"},"tags":[],"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-01-26T22:19:00.291217Z","iopub.execute_input":"2024-01-26T22:19:00.291603Z","iopub.status.idle":"2024-01-26T22:19:00.297192Z","shell.execute_reply.started":"2024-01-26T22:19:00.291569Z","shell.execute_reply":"2024-01-26T22:19:00.296386Z"},"trusted":true},"execution_count":null,"outputs":[]}]}