{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\n### Briefly - downsample and make quick experiments\n\nHere we will downsample the data. I.e. select 10% of data as a kind of \"Playground\" - for quick experiments. \nAnd will train/tune different models on that playground first.\n\n\n### About CV scheme\n\nPublic and private test sets are quite different. (Private - contains new DAY (and donor), while public ONLY new donor).\nSo one should be careful with the validation schemes.\nSome proposals are described here: \n\nNotebook: https://www.kaggle.com/code/alexandervc/mmscel-crossvalidation-schemes\nTopic: https://www.kaggle.com/competitions/open-problems-multimodal/discussion/358860\n\nWe will be based on them. \n\n### Start with selection of \"Playground\" - 10% part of data - to quickly test models, ideas\n\nData is quite big, and so training models, tuning params might take long time. That is not always affordable.\nIn the present script we first choose some 10% part of data - to make quick experiments.\nThat part is chosen to have SAME proporitions of key characteristics: donors, days, cell types as the initial data\n\nSo we can create CV scheme for that \"playground\" part.\nBut we can also simplify even further - for start - use  not 6-fold scheme but just splite by days. And have only 1 train subset for quick experiments. \n\nTo simplify even further we can start play with just one target only. \n\n","metadata":{}},{"cell_type":"markdown","source":"# Install/import modules, load technical data\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-31T17:14:33.409625Z","iopub.execute_input":"2022-10-31T17:14:33.410288Z","iopub.status.idle":"2022-10-31T17:14:33.430162Z","shell.execute_reply.started":"2022-10-31T17:14:33.410250Z","shell.execute_reply":"2022-10-31T17:14:33.428573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nt0start = time.time()\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport sys\n\nimport matplotlib.pyplot as plt\n#plt.style.use('dark_background')\nimport seaborn as sns\n\n#If you see a urllib warning running this cell, go to \"Settings\" on the right hand side, \n#and turn on internet. Note, you need to be phone verified.\n!pip install --quiet tables\n\n\nimport h5py\n!pip install hdf5plugin~=2.0 # https://forum.hdfgroup.org/t/cant-open-directory-usr-local-hdf5-lib-plugin/9738/4\nimport hdf5plugin\n\n# !pip install scanpy\n# import scanpy as sc\n# import anndata\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")\n\ndf_cell = pd.read_csv(FP_CELL_METADATA)\ndf_cell","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:14:33.444400Z","iopub.execute_input":"2022-10-31T17:14:33.444764Z","iopub.status.idle":"2022-10-31T17:14:56.605742Z","shell.execute_reply.started":"2022-10-31T17:14:33.444733Z","shell.execute_reply":"2022-10-31T17:14:56.603463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nt0start = time.time()\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport sys\n\nimport matplotlib.pyplot as plt\n#plt.style.use('dark_background')\nimport seaborn as sns\n\n#If you see a urllib warning running this cell, go to \"Settings\" on the right hand side, \n#and turn on internet. Note, you need to be phone verified.\n!pip install --quiet tables\n\n\nimport h5py\n!pip install hdf5plugin~=2.0 # https://forum.hdfgroup.org/t/cant-open-directory-usr-local-hdf5-lib-plugin/9738/4\nimport hdf5plugin\n\n# !pip install scanpy\n# import scanpy as sc\n# import anndata\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_CELL_METADATA = os.path.join(DATA_DIR,\"metadata.csv\")\n\nFP_CITE_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_cite_inputs.h5\")\nFP_CITE_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_cite_targets.h5\")\nFP_CITE_TEST_INPUTS = os.path.join(DATA_DIR,\"test_cite_inputs.h5\")\n\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nFP_MULTIOME_TEST_INPUTS = os.path.join(DATA_DIR,\"test_multi_inputs.h5\")\n\nFP_SUBMISSION = os.path.join(DATA_DIR,\"sample_submission.csv\")\nFP_EVALUATION_IDS = os.path.join(DATA_DIR,\"evaluation_ids.csv\")\n\ndf_cell = pd.read_csv(FP_CELL_METADATA)\ndf_cell","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:14:56.609536Z","iopub.execute_input":"2022-10-31T17:14:56.610608Z","iopub.status.idle":"2022-10-31T17:15:14.551945Z","shell.execute_reply.started":"2022-10-31T17:14:56.610558Z","shell.execute_reply":"2022-10-31T17:15:14.550570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load prepared Features for CITE-seq part of task","metadata":{}},{"cell_type":"code","source":"%%time\n\nprint('Load prepared features for CITE-seq')\n# These files contain both train and test parts .\n# For CITEseq part - first 70988 elements - train, and later 48663 - test. Overall 119651 samples.\nfn = '/kaggle/input/feature-shop-for-multimodal-singlecell-competition/citeseq_train_and_test_TruncatedSVD200_niter7_rs42.csv'\nfn = '/kaggle/input/feature-shop-for-multimodal-singlecell-competition/citeseq_train_and_test_PCA500.csv'\ndf_cite = pd.read_csv(fn,index_col = 0)\ndisplay(df_cite)\n","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:15:14.554550Z","iopub.execute_input":"2022-10-31T17:15:14.555017Z","iopub.status.idle":"2022-10-31T17:15:31.248093Z","shell.execute_reply.started":"2022-10-31T17:15:14.554975Z","shell.execute_reply":"2022-10-31T17:15:31.247243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cite.mean(axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:15:31.250048Z","iopub.execute_input":"2022-10-31T17:15:31.250322Z","iopub.status.idle":"2022-10-31T17:15:31.358412Z","shell.execute_reply.started":"2022-10-31T17:15:31.250297Z","shell.execute_reply":"2022-10-31T17:15:31.357551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_cite.std(axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:15:31.359506Z","iopub.execute_input":"2022-10-31T17:15:31.359741Z","iopub.status.idle":"2022-10-31T17:15:31.725256Z","shell.execute_reply.started":"2022-10-31T17:15:31.359718Z","shell.execute_reply":"2022-10-31T17:15:31.724146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Targets for CITE-seq","metadata":{}},{"cell_type":"code","source":"%%time\n#if 1:\nprint('Load CITE-seq targets and ')\ndf_cite_train_y = pd.read_hdf(FP_CITE_TRAIN_TARGETS)\ndisplay(df_cite_train_y)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:15:31.726741Z","iopub.execute_input":"2022-10-31T17:15:31.726982Z","iopub.status.idle":"2022-10-31T17:15:32.489196Z","shell.execute_reply.started":"2022-10-31T17:15:31.726959Z","shell.execute_reply":"2022-10-31T17:15:32.487893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load and start prepare some metadata (cut only CITE-seq train part)","metadata":{}},{"cell_type":"code","source":"%%time\nfn2 = '/kaggle/input/feature-shop-for-multimodal-singlecell-competition/_citeseq_meta_all_text_also.csv'\ndf_meta_full = pd.read_csv(fn2,index_col = 0)\ndisplay(df_meta_full)\n#if 1:\ndf_meta = pd.DataFrame(index = df_cite_train_y.index) \ndf_meta = df_meta.join(df_cell.set_index('cell_id') )\ndisplay(df_meta)","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:15:32.491059Z","iopub.execute_input":"2022-10-31T17:15:32.491660Z","iopub.status.idle":"2022-10-31T17:15:32.846774Z","shell.execute_reply.started":"2022-10-31T17:15:32.491597Z","shell.execute_reply":"2022-10-31T17:15:32.845408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create \"Playground\"  (i.e. downsample)\n\nSmall 10% of data of data where we can make prelimanary experiments. \n\nit will be labeled by special column \"Playground\" in df_meta (meta data for CITE-seq train only)","metadata":{}},{"cell_type":"code","source":"# Prepare for creation of additional holdout folds with 10% of samples \n# We will use stratified Kfold to achieve that days, cell_types and donors are equally distributed \nimport numpy as np\nfrom sklearn.model_selection import StratifiedKFold\nscol = 'donor&day&CT'\ndf_meta[scol] =df_meta['donor'].apply(lambda x:str(x)+'_') + df_meta['day'].apply(lambda x:str(x)+'_') + df_meta['cell_type']\n\n\nskf = StratifiedKFold(n_splits=10,  shuffle=True, random_state=40)\nskf.get_n_splits(df_meta, df_meta[scol] )\n\n\n\ny = df_meta[scol] \nfor train_index, test_index in skf.split(df_meta, df_meta[scol]):\n    print(\"TRAIN:\", len(train_index), \"TEST:\", len(test_index) ); \n    break\nprint(test_index)\nprint(df_meta[scol].value_counts().head(5)   )\nprint(df_meta.iloc[test_index,:][scol].value_counts().head(5)    )\n\n\nflagged_column_name = 'Playground'\ndf_meta[flagged_column_name] = 0 \ndf_meta.loc[df_meta.index[test_index],flagged_column_name]  = 1\ndf_meta","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:15:32.848414Z","iopub.execute_input":"2022-10-31T17:15:32.848798Z","iopub.status.idle":"2022-10-31T17:15:33.152149Z","shell.execute_reply.started":"2022-10-31T17:15:32.848763Z","shell.execute_reply":"2022-10-31T17:15:33.151479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create X,y, X_train, y_train, etc - INSIDE \"Playground\"\n","metadata":{}},{"cell_type":"code","source":"n_features = 100\n\nX= df_cite.iloc[:70988,:n_features][df_meta['Playground']==1]\ny= df_cite_train_y[df_meta['Playground']==1]\nprint('X.shape, y.shape', X.shape, y.shape )\n\n# Create simplfied validation scheme - like real test data - with two test-sets private-like, public-like:\n# Private like test - new DAY, and donor, \n# While public like - only new donor (days are the same as in train):\n# Step 1: \nmask_train = (df_meta['Playground']==1)&(df_meta['day']!=4)&(df_meta['donor']!=31800) \nX_train = df_cite.iloc[:70988,:n_features][mask_train]\ny_train = df_cite_train_y[mask_train]\n\n# Step 2:\nmask_test_private_like = (df_meta['Playground']==1)&(df_meta['day']==4)\nX_test_private_like = df_cite.iloc[:70988,:n_features][ mask_test_private_like  ]\ny_test_private_like = df_cite_train_y[mask_test_private_like]\nX_test = X_test_private_like\ny_test = y_test_private_like\n# Step 3: \nmask_test_public_like = (df_meta['Playground']==1)&(df_meta['day']!=4)  &(df_meta['donor']==31800) \nX_test_public_like = df_cite.iloc[:70988,:n_features][mask_test_public_like ]\ny_test_public_like = df_cite_train_y[mask_test_public_like]\n\nX.shape,y.shape, X_train.shape, X_test_private_like.shape, X_test_public_like.shape, y_train.shape, y_test_private_like.shape, y_test_public_like.shape\n\n","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:15:33.153318Z","iopub.execute_input":"2022-10-31T17:15:33.153598Z","iopub.status.idle":"2022-10-31T17:15:33.239673Z","shell.execute_reply.started":"2022-10-31T17:15:33.153571Z","shell.execute_reply":"2022-10-31T17:15:33.238238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score\nimport lightgbm as lgbm","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:15:33.245619Z","iopub.execute_input":"2022-10-31T17:15:33.246000Z","iopub.status.idle":"2022-10-31T17:15:34.203392Z","shell.execute_reply.started":"2022-10-31T17:15:33.245965Z","shell.execute_reply":"2022-10-31T17:15:34.202423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LightGBM model","metadata":{}},{"cell_type":"markdown","source":"# Finding params","metadata":{}},{"cell_type":"code","source":"import optuna\nimport warnings\nwarnings.simplefilter('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:20:29.405591Z","iopub.execute_input":"2022-10-31T17:20:29.405937Z","iopub.status.idle":"2022-10-31T17:20:29.411583Z","shell.execute_reply.started":"2022-10-31T17:20:29.405907Z","shell.execute_reply":"2022-10-31T17:20:29.410294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_for_tuning = [\"CD31\", \"CD44\"]\nfeatures_for_tuning = {f:[] for f in features_for_tuning}","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:59:09.824384Z","iopub.execute_input":"2022-10-31T17:59:09.824848Z","iopub.status.idle":"2022-10-31T17:59:09.832865Z","shell.execute_reply.started":"2022-10-31T17:59:09.824811Z","shell.execute_reply":"2022-10-31T17:59:09.831029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fixed_params = {      \n    'random_state': 42,\n    'min_split_gain': 12.2,\n    'subsample_for_bin': 10000, \n    'other_rate': 1,\n    'reg_alpha': 0.0,\n    'reg_lambda': 0.0, \n#     'min_child_weight': 0.1, \n#     'learning_rate': 0.07 ,  # hypersensitive - change to 0.0671 - worsens about 2% from 0.43590181064083766\n#     'num_leaves': 12, # hypersensitive - change to 11, worsens about 2%\n#     'max_depth': 9, # hypersensitive - change to 10 - worsens about 1%\n#     'n_estimators': 150, # senstitive - change by 10 - worsents about 0.5%\n#     'min_child_samples': 20,# hypersensitive - change to 21 - worsens about 3% \n#     'min_split_gain': 12.2,# sometimes hypersenstive - change by 0.1 worsens by 0.9%, But for 100 features changes 11.8-12.2 - not change AT ALLL !!!  # Uplifts to  0.43590.. !!! \n#     'reg_lambda': 0.0, # seems only makes worse\n#     'reg_alpha': 0.0, # seems only makes worse\n#     'subsample_for_bin': 10000, # seems  less 10000 - worsens, but after 10 000 does not influence  \n#     'colsample_bytree': 1, # only worsens\n#     'subsample': 1, # Does not seem to influence at all\n#     'other_rate': 1,# no influence ?  \n#     'min_child_weight': 0.1, # no influence ?\n#     'subsample_freq':10,# no influence ? # Integer; alias: bagging_freq; k means perform bagging at every k iteration\n}\n\ndef objective_main(trial, X_train, y_train, X_test, y_test):\n    params = {\n        'metric': 'r2', \n        **fixed_params,\n        'n_estimators': trial.suggest_int('n_estimators', 10, 1000),\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n        'max_depth': trial.suggest_int('max_depth', 5, 30),\n        'num_leaves' : trial.suggest_int('num_leaves', 5, 50),\n        'min_child_samples': trial.suggest_int('min_child_samples', 1, 50),\n        'min_child_weight': trial.suggest_loguniform('min_child_weight', 0.1, 1.0),        \n        'colsample_bytree': trial.suggest_categorical('colsample_bytree', [1.0,0.2,0.4,0.6,0.8]),\n        'subsample': trial.suggest_categorical('subsample', [0.4,0.6,0.8,1.0]),\n        'subsample_freq' : trial.suggest_int('subsample_freq', 1, 50),\n        'cat_smooth' : trial.suggest_int('cat_smooth', 1, 100), \n#         'reg_alpha': trial.suggest_loguniform('reg_alpha', 0.00001, 1.0),        \n#         'reg_lambda': trial.suggest_loguniform('reg_lambda', 0.00001, 1.0),        \n\n    }\n    model = lgbm.LGBMRegressor(**params)\n\n    model.fit(X_train, y_train)\n\n    y_pred = model.predict(X_test)\n    mse = mean_squared_error(y_pred, y_test)\n    return mse\n","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:59:10.114723Z","iopub.execute_input":"2022-10-31T17:59:10.115243Z","iopub.status.idle":"2022-10-31T17:59:10.128410Z","shell.execute_reply.started":"2022-10-31T17:59:10.115165Z","shell.execute_reply":"2022-10-31T17:59:10.127053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor feature in features_for_tuning:\n    feature_pos = df_cite_train_y.columns.get_loc(feature)\n    objective = lambda trial: objective_main(trial, \n                                             X_train, y_train[feature], \n                                             X_test, y_test[feature])\n    study = optuna.create_study(\n        direction='minimize', \n        pruner=optuna.pruners.MedianPruner(n_warmup_steps=50),\n        study_name=feature)\n    study.optimize(objective, n_trials=50)\n    features_for_tuning[feature] = {**fixed_params, **study.best_trial.params}\n    print('Best trial: ', study.best_trial.number, study.best_trial.params)\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:59:10.493013Z","iopub.execute_input":"2022-10-31T17:59:10.494602Z","iopub.status.idle":"2022-10-31T18:01:24.261500Z","shell.execute_reply.started":"2022-10-31T17:59:10.494544Z","shell.execute_reply":"2022-10-31T18:01:24.260702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_for_tuning","metadata":{"execution":{"iopub.status.busy":"2022-10-31T18:01:24.265032Z","iopub.execute_input":"2022-10-31T18:01:24.265511Z","iopub.status.idle":"2022-10-31T18:01:24.275505Z","shell.execute_reply.started":"2022-10-31T18:01:24.265476Z","shell.execute_reply":"2022-10-31T18:01:24.274416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor feature in features_for_tuning:\n    model = lgbm.LGBMRegressor(**features_for_tuning[feature])\n    model.fit(X_train,y_train[feature])\n    y_pred = model.predict(X_test)\n    print(f'{feature}: r2 - {round(r2_score(y_test[feature], y_pred), 5)}, mse - {round(mean_squared_error(y_test[feature], y_pred), 5)}')","metadata":{"execution":{"iopub.status.busy":"2022-10-31T18:02:58.539237Z","iopub.execute_input":"2022-10-31T18:02:58.539725Z","iopub.status.idle":"2022-10-31T18:02:59.529629Z","shell.execute_reply.started":"2022-10-31T18:02:58.539679Z","shell.execute_reply":"2022-10-31T18:02:59.528533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dir(model)\nmodel.get_params()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T18:01:25.178557Z","iopub.execute_input":"2022-10-31T18:01:25.179030Z","iopub.status.idle":"2022-10-31T18:01:25.196253Z","shell.execute_reply.started":"2022-10-31T18:01:25.178997Z","shell.execute_reply":"2022-10-31T18:01:25.194411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ridge","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\n","metadata":{"execution":{"iopub.status.busy":"2022-10-31T18:03:13.946297Z","iopub.execute_input":"2022-10-31T18:03:13.946679Z","iopub.status.idle":"2022-10-31T18:03:13.950945Z","shell.execute_reply.started":"2022-10-31T18:03:13.946648Z","shell.execute_reply":"2022-10-31T18:03:13.950081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor feature in features_for_tuning:\n    model = Ridge(alpha=1)\n    model.fit(X_train,y_train[feature])\n    y_pred = model.predict(X_test)\n    print(f'{feature}: r2 - {round(r2_score(y_test[feature], y_pred), 5)}, mse - {round(mean_squared_error(y_test[feature], y_pred), 5)}')","metadata":{"execution":{"iopub.status.busy":"2022-10-31T18:03:14.929670Z","iopub.execute_input":"2022-10-31T18:03:14.930177Z","iopub.status.idle":"2022-10-31T18:03:14.983655Z","shell.execute_reply.started":"2022-10-31T18:03:14.930132Z","shell.execute_reply":"2022-10-31T18:03:14.982845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor feature in features_for_tuning:\n    model = Ridge(alpha=1e4) \n    model.fit(X_train,y_train[feature])\n    y_pred = model.predict(X_test)\n    print(f'{feature}: r2 - {round(r2_score(y_test[feature], y_pred), 5)}, mse - {round(mean_squared_error(y_test[feature], y_pred), 5)}')","metadata":{"execution":{"iopub.status.busy":"2022-10-31T18:03:19.251005Z","iopub.execute_input":"2022-10-31T18:03:19.251455Z","iopub.status.idle":"2022-10-31T18:03:19.296272Z","shell.execute_reply.started":"2022-10-31T18:03:19.251421Z","shell.execute_reply":"2022-10-31T18:03:19.295330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor feature in features_for_tuning:\n    model = Ridge(alpha=1e5) # lgbm.LGBMRegressor()#**params)\n    model.fit(X_train,y_train[feature])\n    y_pred = model.predict(X_test)\n    print(f'{feature}: r2 - {round(r2_score(y_test[feature], y_pred), 5)}, mse - {round(mean_squared_error(y_test[feature], y_pred), 5)}')","metadata":{"execution":{"iopub.status.busy":"2022-10-31T18:03:23.096628Z","iopub.execute_input":"2022-10-31T18:03:23.097077Z","iopub.status.idle":"2022-10-31T18:03:23.152148Z","shell.execute_reply.started":"2022-10-31T18:03:23.097033Z","shell.execute_reply":"2022-10-31T18:03:23.151389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KernelRidge","metadata":{}},{"cell_type":"code","source":"%%time\nfrom sklearn.kernel_ridge import KernelRidge\nfor feature in features_for_tuning:\n\n    model = KernelRidge(alpha=1.0)\n    model.fit(X_train,y_train[feature])\n    y_pred = model.predict(X_test)\n    print(r2_score(y_test[feature], y_pred), mean_squared_error(y_test[feature], y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:15:48.380944Z","iopub.execute_input":"2022-10-31T17:15:48.383431Z","iopub.status.idle":"2022-10-31T17:15:48.780920Z","shell.execute_reply.started":"2022-10-31T17:15:48.383394Z","shell.execute_reply":"2022-10-31T17:15:48.778734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"execution":{"iopub.status.busy":"2022-10-31T17:15:48.782198Z","iopub.execute_input":"2022-10-31T17:15:48.782558Z","iopub.status.idle":"2022-10-31T17:15:48.789012Z","shell.execute_reply.started":"2022-10-31T17:15:48.782521Z","shell.execute_reply":"2022-10-31T17:15:48.787067Z"},"trusted":true},"execution_count":null,"outputs":[]}]}