{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\n### Briefly - downsample and make quick experiments\n\nHere we will downsample the data. I.e. select 10% of data as a kind of \"Playground\" - for quick experiments. \nAnd will train/tune different models on that playground first.\n\n\n### About CV scheme\n\nPublic and private test sets are quite different. (Private - contains new DAY (and donor), while public ONLY new donor).\nSo one should be careful with the validation schemes.\nSome proposals are described here: \n\nNotebook: https://www.kaggle.com/code/alexandervc/mmscel-crossvalidation-schemes\nTopic: https://www.kaggle.com/competitions/open-problems-multimodal/discussion/358860\n\nWe will be based on them. \n\n### Start with selection of \"Playground\" - 10% part of data - to quickly test models, ideas\n\nData is quite big, and so training models, tuning params might take long time. That is not always affordable.\nIn the present script we first choose some 10% part of data - to make quick experiments.\nThat part is chosen to have SAME proporitions of key characteristics: donors, days, cell types as the initial data\n\nSo we can create CV scheme for that \"playground\" part.\nBut we can also simplify even further - for start - use  not 6-fold scheme but just splite by days. And have only 1 train subset for quick experiments. \n\nTo simplify even further we can start play with just one target only. \n\n\n\n### Versions\n\n\n#### 1 Test several models - Ridge,LGB, SVR, RF etc...\n\n    Target chosen - only CD31\n    No params tuning - just fist look \n","metadata":{}},{"cell_type":"markdown","source":"# Install/import modules, load technical data\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-29T10:07:57.227667Z","iopub.execute_input":"2022-10-29T10:07:57.228062Z","iopub.status.idle":"2022-10-29T10:07:57.266312Z","shell.execute_reply.started":"2022-10-29T10:07:57.228033Z","shell.execute_reply":"2022-10-29T10:07:57.264991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nt0start = time.time()\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport sys\n\nimport matplotlib.pyplot as plt\n#plt.style.use('dark_background')\nimport seaborn as sns\n\n#If you see a urllib warning running this cell, go to \"Settings\" on the right hand side, \n#and turn on internet. Note, you need to be phone verified.\n!pip install --quiet tables\n\n\nimport h5py\n!pip install hdf5plugin~=2.0 # https://forum.hdfgroup.org/t/cant-open-directory-usr-local-hdf5-lib-plugin/9738/4\nimport hdf5plugin\n\n# !pip install scanpy\n# import scanpy as sc\n# import anndata\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")\n\ndf_cell = pd.read_csv(FP_CELL_METADATA)\ndf_cell","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:07:57.432174Z","iopub.execute_input":"2022-10-29T10:07:57.433332Z","iopub.status.idle":"2022-10-29T10:08:23.773842Z","shell.execute_reply.started":"2022-10-29T10:07:57.433244Z","shell.execute_reply":"2022-10-29T10:08:23.772271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nt0start = time.time()\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport sys\n\nimport matplotlib.pyplot as plt\n#plt.style.use('dark_background')\nimport seaborn as sns\n\n#If you see a urllib warning running this cell, go to \"Settings\" on the right hand side, \n#and turn on internet. Note, you need to be phone verified.\n!pip install --quiet tables\n\n\nimport h5py\n!pip install hdf5plugin~=2.0 # https://forum.hdfgroup.org/t/cant-open-directory-usr-local-hdf5-lib-plugin/9738/4\nimport hdf5plugin\n\n# !pip install scanpy\n# import scanpy as sc\n# import anndata\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")\n\ndf_cell = pd.read_csv(FP_CELL_METADATA)\ndf_cell","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:08:23.776641Z","iopub.execute_input":"2022-10-29T10:08:23.777773Z","iopub.status.idle":"2022-10-29T10:08:46.435715Z","shell.execute_reply.started":"2022-10-29T10:08:23.777732Z","shell.execute_reply":"2022-10-29T10:08:46.434121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load prepared Features for CITE-seq part of task","metadata":{}},{"cell_type":"code","source":"%%time\n\nprint('Load prepared features for CITE-seq')\n# These files contain both train and test parts .\n# For CITEseq part - first 70988 elements - train, and later 48663 - test. Overall 119651 samples.\nfn = '/kaggle/input/feature-shop-for-multimodal-singlecell-competition/citeseq_train_and_test_TruncatedSVD200_niter7_rs42.csv'\nfn = '/kaggle/input/feature-shop-for-multimodal-singlecell-competition/citeseq_train_and_test_PCA500.csv'\ndf_cite = pd.read_csv(fn,index_col = 0)\ndisplay(df_cite)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:08:46.437453Z","iopub.execute_input":"2022-10-29T10:08:46.437938Z","iopub.status.idle":"2022-10-29T10:09:08.314727Z","shell.execute_reply.started":"2022-10-29T10:08:46.437904Z","shell.execute_reply":"2022-10-29T10:09:08.313347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cite.mean(axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:09:08.318069Z","iopub.execute_input":"2022-10-29T10:09:08.318837Z","iopub.status.idle":"2022-10-29T10:09:08.497590Z","shell.execute_reply.started":"2022-10-29T10:09:08.318788Z","shell.execute_reply":"2022-10-29T10:09:08.496635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cite.std(axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:09:08.498963Z","iopub.execute_input":"2022-10-29T10:09:08.499580Z","iopub.status.idle":"2022-10-29T10:09:09.224759Z","shell.execute_reply.started":"2022-10-29T10:09:08.499544Z","shell.execute_reply":"2022-10-29T10:09:09.223559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Targets for CITE-seq","metadata":{}},{"cell_type":"code","source":"%%time\n#if 1:\nprint('Load CITE-seq targets and ')\ndf_cite_train_y = pd.read_hdf(FP_CITE_TRAIN_TARGETS)\ndisplay(df_cite_train_y)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:09:09.226652Z","iopub.execute_input":"2022-10-29T10:09:09.227424Z","iopub.status.idle":"2022-10-29T10:09:10.031874Z","shell.execute_reply.started":"2022-10-29T10:09:09.227377Z","shell.execute_reply":"2022-10-29T10:09:10.030655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load and start prepare some metadata (cut only CITE-seq train part)","metadata":{}},{"cell_type":"code","source":"%%time\nfn2 = '/kaggle/input/feature-shop-for-multimodal-singlecell-competition/_citeseq_meta_all_text_also.csv'\ndf_meta_full = pd.read_csv(fn2,index_col = 0)\ndisplay(df_meta_full)\n#if 1:\ndf_meta = pd.DataFrame(index = df_cite_train_y.index) \ndf_meta = df_meta.join(df_cell.set_index('cell_id') )\ndisplay(df_meta)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:09:10.033408Z","iopub.execute_input":"2022-10-29T10:09:10.033769Z","iopub.status.idle":"2022-10-29T10:09:10.430963Z","shell.execute_reply.started":"2022-10-29T10:09:10.033736Z","shell.execute_reply":"2022-10-29T10:09:10.430011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create \"Playground\"  (i.e. downsample)\n\nSmall 10% of data of data where we can make prelimanary experiments. \n\nit will be labeled by special column \"Playground\" in df_meta (meta data for CITE-seq train only)","metadata":{}},{"cell_type":"code","source":"# Prepare for creation of additional holdout folds with 10% of samples \n# We will use stratified Kfold to achieve that days, cell_types and donors are equally distributed \nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nscol = 'donor&day&CT'\ndf_meta[scol] =df_meta['donor'].apply(lambda x:str(x)+'_') + df_meta['day'].apply(lambda x:str(x)+'_') + df_meta['cell_type']\n\n\nskf = StratifiedKFold(n_splits=10,  shuffle=True, random_state=40)\nskf.get_n_splits(df_meta, df_meta[scol] )\n\n\n\ny = df_meta[scol] \nfor train_index, test_index in skf.split(df_meta, df_meta[scol]):\n    print(\"TRAIN:\", len(train_index), \"TEST:\", len(test_index) ); \n    break\nprint(test_index)\nprint(df_meta[scol].value_counts().head(5)   )\nprint(df_meta.iloc[test_index,:][scol].value_counts().head(5)    )\n\n\nflagged_column_name = 'Playground'\ndf_meta[flagged_column_name] = 0 \ndf_meta.loc[df_meta.index[test_index],flagged_column_name]  = 1\ndf_meta","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:09:10.432267Z","iopub.execute_input":"2022-10-29T10:09:10.432948Z","iopub.status.idle":"2022-10-29T10:09:10.668278Z","shell.execute_reply.started":"2022-10-29T10:09:10.432912Z","shell.execute_reply":"2022-10-29T10:09:10.666906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create X,y, X_train, y_train, etc - INSIDE \"Playground\"\n","metadata":{}},{"cell_type":"code","source":"selected_target = 'CD31'\n\nX= df_cite.iloc[:70988,:][df_meta['Playground']==1]\ny= df_cite_train_y[df_meta['Playground']==1][selected_target]\nprint('X.shape, y.shape', X.shape, y.shape )\n\n# Create simplfied validation scheme - like real test data - with two test-sets private-like, public-like:\n# Private like test - new DAY, and donor, \n# While public like - only new donor (days are the same as in train):\n# Step 1: \nmask_train = (df_meta['Playground']==1)&(df_meta['day']!=4)&(df_meta['donor']!=31800) \nX_train = df_cite.iloc[:70988,:][mask_train]\ny_train = df_cite_train_y[mask_train][ selected_target ]\n# Step 2:\nmask_test_private_like = (df_meta['Playground']==1)&(df_meta['day']==4)\nX_test_private_like = df_cite.iloc[:70988,:][ mask_test_private_like  ]\ny_test_private_like = df_cite_train_y[mask_test_private_like][ selected_target ]\nX_test = X_test_private_like\ny_test = y_test_private_like\n# Step 3: \nmask_test_public_like = (df_meta['Playground']==1)&(df_meta['day']!=4)  &(df_meta['donor']==31800) \nX_test_public_like = df_cite.iloc[:70988,:][mask_test_public_like ]\ny_test_public_like = df_cite_train_y[mask_test_public_like][ selected_target ]\n\nX.shape,y.shape, X_train.shape, X_test_private_like.shape, X_test_public_like.shape, y_train.shape, y_test_private_like.shape, y_test_public_like.shape\n\n","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:09:10.670238Z","iopub.execute_input":"2022-10-29T10:09:10.670950Z","iopub.status.idle":"2022-10-29T10:09:10.824105Z","shell.execute_reply.started":"2022-10-29T10:09:10.670914Z","shell.execute_reply":"2022-10-29T10:09:10.823104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LightGBM model","metadata":{}},{"cell_type":"code","source":"%%time\nimport lightgbm as lgbm\nfrom hyperopt import hp, tpe, Trials\nfrom hyperopt.fmin import fmin\nimport hyperopt\nfrom sklearn.model_selection import KFold","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:09:24.442276Z","iopub.execute_input":"2022-10-29T10:09:24.442702Z","iopub.status.idle":"2022-10-29T10:09:25.667647Z","shell.execute_reply.started":"2022-10-29T10:09:24.442669Z","shell.execute_reply":"2022-10-29T10:09:25.666337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel = lgbm.LGBMRegressor()#**params)\nmodel.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:09:25.895603Z","iopub.execute_input":"2022-10-29T10:09:25.895993Z","iopub.status.idle":"2022-10-29T10:09:30.869541Z","shell.execute_reply.started":"2022-10-29T10:09:25.895962Z","shell.execute_reply":"2022-10-29T10:09:30.868457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel = lgbm.LGBMRegressor(num_leaves=16)#**params)\nmodel.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:35:29.778083Z","iopub.execute_input":"2022-10-29T10:35:29.778787Z","iopub.status.idle":"2022-10-29T10:35:32.185576Z","shell.execute_reply.started":"2022-10-29T10:35:29.778749Z","shell.execute_reply":"2022-10-29T10:35:32.184585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:09:41.154030Z","iopub.execute_input":"2022-10-29T10:09:41.154486Z","iopub.status.idle":"2022-10-29T10:09:41.159752Z","shell.execute_reply.started":"2022-10-29T10:09:41.154446Z","shell.execute_reply":"2022-10-29T10:09:41.158846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_test)\nr2_score(y_test, y_pred), mean_squared_error(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:35:32.189951Z","iopub.execute_input":"2022-10-29T10:35:32.190682Z","iopub.status.idle":"2022-10-29T10:35:32.227490Z","shell.execute_reply.started":"2022-10-29T10:35:32.190643Z","shell.execute_reply":"2022-10-29T10:35:32.226530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape, X_test_private_like.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:29:39.447232Z","iopub.execute_input":"2022-10-29T10:29:39.447702Z","iopub.status.idle":"2022-10-29T10:29:39.454332Z","shell.execute_reply.started":"2022-10-29T10:29:39.447662Z","shell.execute_reply":"2022-10-29T10:29:39.453515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_private_like","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:29:54.720442Z","iopub.execute_input":"2022-10-29T10:29:54.720887Z","iopub.status.idle":"2022-10-29T10:29:54.731228Z","shell.execute_reply.started":"2022-10-29T10:29:54.720848Z","shell.execute_reply":"2022-10-29T10:29:54.729938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:30:06.154051Z","iopub.execute_input":"2022-10-29T10:30:06.154473Z","iopub.status.idle":"2022-10-29T10:30:06.189765Z","shell.execute_reply.started":"2022-10-29T10:30:06.154438Z","shell.execute_reply":"2022-10-29T10:30:06.188462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_private_like","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:30:13.925449Z","iopub.execute_input":"2022-10-29T10:30:13.925859Z","iopub.status.idle":"2022-10-29T10:30:13.959693Z","shell.execute_reply.started":"2022-10-29T10:30:13.925827Z","shell.execute_reply":"2022-10-29T10:30:13.958494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:29:48.066525Z","iopub.execute_input":"2022-10-29T10:29:48.067828Z","iopub.status.idle":"2022-10-29T10:29:48.076794Z","shell.execute_reply.started":"2022-10-29T10:29:48.067782Z","shell.execute_reply":"2022-10-29T10:29:48.075340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_test_private_like)\nr2_score(y_test_private_like, y_pred), mean_squared_error(y_test_private_like, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:30:36.185236Z","iopub.execute_input":"2022-10-29T10:30:36.185677Z","iopub.status.idle":"2022-10-29T10:30:36.223676Z","shell.execute_reply.started":"2022-10-29T10:30:36.185639Z","shell.execute_reply":"2022-10-29T10:30:36.222691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:30:36.433115Z","iopub.execute_input":"2022-10-29T10:30:36.434582Z","iopub.status.idle":"2022-10-29T10:30:36.442653Z","shell.execute_reply.started":"2022-10-29T10:30:36.434483Z","shell.execute_reply":"2022-10-29T10:30:36.441697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_test_public_like)\nr2_score(y_test_public_like, y_pred), mean_squared_error(y_test_public_like, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T10:21:35.587506Z","iopub.execute_input":"2022-10-29T10:21:35.587946Z","iopub.status.idle":"2022-10-29T10:21:35.620574Z","shell.execute_reply.started":"2022-10-29T10:21:35.587915Z","shell.execute_reply":"2022-10-29T10:21:35.619364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check if given parameter can be interpreted as a numerical value\ndef is_number(s):\n    if s is None:\n        return False\n    try:\n        float(s)\n        return True\n    except ValueError:\n        return False\n\n#convert given set of paramaters to integer values\n#this at least cuts the excess float decimals if they are there\ndef convert_int_params(names, params):\n    for int_type in names:\n        #sometimes the parameters can be choices between options or numerical values. like \"log2\" vs \"1-10\"\n        raw_val = params[int_type]\n        if is_number(raw_val):\n            params[int_type] = int(raw_val)\n    return params\n\n#convert float parameters to 3 digit precision strings\n#just for simpler diplay and all\ndef convert_float_params(names, params):\n    for float_type in names:\n        raw_val = params[float_type]\n        if is_number(raw_val):\n            params[float_type] = '{:.3f}'.format(raw_val)\n    return params\n\n\n# how many CV folds to do on the data\nn_folds = 5\n# max number of rows to use for X and y. to reduce time and compare options faster\nmax_n = None\n# max number of trials hyperopt runs\nn_trials = 200\n#verbosity in LGBM is how often progress is printed. with 100=print progress every 100 rounds. 0 is quite?\nverbosity = False\nprint_summary = False\n\nall_scores = []\nall_params = []\n\n# run n_folds of cross validation on the data\n# averages fold results\ndef fit_cv(X, y, params, fit_params):\n    # cut the data if max_n is set\n    if max_n is not None:\n        X = X[:max_n]\n        y = y[:max_n]\n    \n    y = np.array(y)\n\n    score = 0\n    folds = KFold(n_splits=n_folds, shuffle=True, random_state=42)\n\n    if print_summary:\n        print(f\"Running {n_folds} folds...\")\n    oof_preds = np.zeros((X.shape[0]))\n    for i, (train_index, test_index) in enumerate(folds.split(X, y)):\n        if verbosity > 0:\n            print('-' * 20, f\"RUNNING FOLD: {i}/{n_folds}\", '-' * 20)\n\n        model = lgbm.LGBMRegressor(**params)\n        X_train, y_train = X.iloc[train_index], y[train_index]\n        X_test, y_test = X.iloc[test_index], y[test_index]\n        #if 100 it prints progress 100,200,300,... iterations\n        model.fit(X_train, y_train, \n                  eval_set=(X_test, y_test), \n                  **fit_params, \n                  callbacks=[lgbm.callback.early_stopping(50, verbose=False)])\n        oof_preds[test_index] = model.predict(X.iloc[test_index])\n        score += mean_squared_error(y[test_index], oof_preds[test_index])\n        importances = model.feature_importances_\n        features = X.columns\n        \n    total_score = score / n_folds\n    all_scores.append(total_score)\n    all_params.append(params)\n    if print_summary:\n        print(f\"total score: {total_score}\")\n    return total_score\n\n\ndef fit_no_cv(params, fit_params):\n    model = lgbm.LGBMRegressor(**params)\n    #if 100 it prints progress 100,200,300,... iterations\n    model.fit(X_train, y_train, \n              eval_set=(X_test_public_like, y_test_public_like), \n              **fit_params, \n              callbacks=[lgbm.callback.early_stopping(50, verbose=False)])\n    oof_preds = model.predict(X_test_private_like)\n    score = mean_squared_error(y_test_private_like, oof_preds)\n    importances = model.feature_importances_\n    features = X.columns\n\n    all_scores.append(score)\n    all_params.append(params)\n    if print_summary:\n        print(f\"score: {score}\")\n    return score\n\ndef create_fit_params(params):\n    using_dart = params['boosting_type'] == \"dart\"\n    fit_params = {\"eval_metric\": \"mse\"}\n    if using_dart:\n        n_estimators = 2000\n    else:\n        n_estimators = 10000\n    params[\"n_estimators\"] = n_estimators\n    return fit_params\n\n\n# this is the objective function the hyperopt aims to minimize\n# i call it objective_sklearn because the lgbm functions called use sklearn API\ndef objective_sklearn(params):\n    int_types = [\"num_leaves\", \"max_depth\", \"min_child_samples\", \n                 \"subsample_for_bin\", \"min_data_in_leaf\", \"bagging_freq\"]\n    params = convert_int_params(int_types, params)\n\n    # Extract the boosting type\n    params['boosting_type'] = params['boosting_type']['boosting_type']\n    #    print(\"running with params:\"+str(params))\n\n    fit_params = create_fit_params(params)\n\n#     score = fit_cv(X, y, params, fit_params)\n    score = fit_no_cv(params, fit_params)\n    if verbosity == 0:\n        if print_summary:\n            print(\"Score {:.3f}\".format(score))\n    else:\n        print(\"Score {:.3f} params {}\".format(score, params))\n    result = {\"loss\": score, \"score\": score, \"params\": params, 'status': hyperopt.STATUS_OK}\n    return result\n\ndef optimize_lgbm(max_n_search=None, boosting_type=None):\n    # https://github.com/Microsoft/LightGBM/blob/master/docs/Parameters.rst\n    # https://indico.cern.ch/event/617754/contributions/2590694/attachments/1459648/2254154/catboost_for_CMS.pdf\n    space = {\n        #this is just piling on most of the possible parameter values for LGBM\n        #some of them apparently don't make sense together, but works for now.. :)\n        'boosting_type': hp.choice('boosting_type',\n                                   [{'boosting_type': 'gbdt',\n                                     }]),\n        'num_leaves': hp.quniform('num_leaves', 12, 201, 12),\n        'max_depth': hp.quniform('max_depth', 4, 8, 1),\n        'learning_rate': hp.uniform('learning_rate', 0.01, 0.2),\n        'subsample_for_bin': hp.quniform('subsample_for_bin', 500, 1501, 100),\n        'feature_fraction': hp.uniform('feature_fraction', 0.8, 1),\n        'feature_fraction_bynode': hp.uniform('feature_fraction_bynode', 0.8, 0.99),\n        'bagging_freq': hp.quniform('bagging_freq', 2, 21, 2),\n        'bagging_fraction': hp.uniform('bagging_fraction', 0.8, 1), #alias \"subsample\"\n        'min_data_in_leaf': hp.quniform('min_data_in_leaf', 10, 271, 30),\n        'lambda_l1': hp.uniform('lambda_l1', 0, 0.7),\n        'lambda_l2': hp.uniform('lambda_l2', 0, 1.2),\n        'seed': 42,\n        #the LGBM parameters docs list various aliases, and the LGBM implementation seems to complain about\n        #the following not being used due to other params, so trying to silence the complaints by setting to None\n        'subsample': None, #overridden by bagging_fraction\n        'subsample_freq': None, #overridden by bagging_freq\n        'reg_alpha': None, #overridden by lambda_l1\n        'reg_lambda': None, #overridden by lambda_l2\n        'min_sum_hessian_in_leaf': None, #overrides min_child_weight\n        'min_child_samples': None, #overridden by min_data_in_leaf\n        'colsample_bytree': None, #overridden by feature_fraction\n#        'min_child_samples': hp.quniform('min_child_samples', 20, 500, 5),\n        'min_child_weight': hp.loguniform('min_child_weight', -2, 3), #also aliases to min_sum_hessian\n        'metric': 'mse',\n    }\n    space['objective'] = \"rmse\"\n    if boosting_type:\n        space['boosting_type'] = boosting_type\n\n    global max_n\n    max_n = max_n_search\n    trials = Trials()\n    best = fmin(fn=objective_sklearn,\n                space=space,\n                algo=tpe.suggest,\n                max_evals=n_trials,\n                trials=trials,\n                verbose= 1)\n\n    # find the trial with lowest loss value. this is what we consider the best one\n    idx = np.argmin(trials.losses())\n    print(idx)\n\n    print(trials.trials[idx])\n\n    # these should be the training parameters to use to achieve the best score in best trial\n    params = trials.trials[idx][\"result\"][\"params\"]\n    max_n = None\n\n    print('==============================')\n    print('= PARAMS')\n    print('==============================')\n    print(params)\n    return params, trials\n\n# run a search\ndef run_lgb(max_n=120000):\n    # the param is the number of rows to use for training\n    params, trials = optimize_lgbm(max_n)\n    print(params)\n\n    return params, trials","metadata":{"execution":{"iopub.status.busy":"2022-10-29T12:04:08.746779Z","iopub.execute_input":"2022-10-29T12:04:08.747221Z","iopub.status.idle":"2022-10-29T12:04:08.782820Z","shell.execute_reply.started":"2022-10-29T12:04:08.747187Z","shell.execute_reply":"2022-10-29T12:04:08.780900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params, trials = optimize_lgbm()\n# boosting_type={\"boosting_type\": \"rf\"}","metadata":{"execution":{"iopub.status.busy":"2022-10-29T12:04:09.254316Z","iopub.execute_input":"2022-10-29T12:04:09.254720Z","iopub.status.idle":"2022-10-29T12:08:40.549447Z","shell.execute_reply.started":"2022-10-29T12:04:09.254686Z","shell.execute_reply":"2022-10-29T12:08:40.547319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel = lgbm.LGBMRegressor(**params)#**params)\nmodel.fit(X_train,y_train)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_test)\nr2_score(y_test, y_pred), mean_squared_error(y_test, y_pred)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"execution":{"iopub.status.busy":"2022-10-28T11:40:31.076516Z","iopub.execute_input":"2022-10-28T11:40:31.077004Z","iopub.status.idle":"2022-10-28T11:40:31.0841Z","shell.execute_reply.started":"2022-10-28T11:40:31.076961Z","shell.execute_reply":"2022-10-28T11:40:31.082408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import BayesianRidge","metadata":{"execution":{"iopub.status.busy":"2022-10-29T12:12:59.195129Z","iopub.execute_input":"2022-10-29T12:12:59.196486Z","iopub.status.idle":"2022-10-29T12:12:59.202379Z","shell.execute_reply.started":"2022-10-29T12:12:59.196431Z","shell.execute_reply":"2022-10-29T12:12:59.201003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel = BayesianRidge()#**params)\nmodel.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T12:15:22.915337Z","iopub.execute_input":"2022-10-29T12:15:22.916005Z","iopub.status.idle":"2022-10-29T12:15:23.285155Z","shell.execute_reply.started":"2022-10-29T12:15:22.915964Z","shell.execute_reply":"2022-10-29T12:15:23.281351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_test)\nr2_score(y_test, y_pred), mean_squared_error(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-10-29T12:15:23.309513Z","iopub.execute_input":"2022-10-29T12:15:23.310235Z","iopub.status.idle":"2022-10-29T12:15:23.355950Z","shell.execute_reply.started":"2022-10-29T12:15:23.310172Z","shell.execute_reply":"2022-10-29T12:15:23.354348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}