{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nAnalyse blend of several solutions \n\n\n### Versions \n\n\n#### 15\n\n    MLP2 models added\n\n#### 13\n    \n    Resnet and LGBweak models added \n    \n#### 12\n    7 more XGB added - total 13 XGB - full average - slighly worse. \n    \n    ","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-13T18:42:23.001854Z","iopub.execute_input":"2022-11-13T18:42:23.003829Z","iopub.status.idle":"2022-11-13T18:42:23.118327Z","shell.execute_reply.started":"2022-11-13T18:42:23.003758Z","shell.execute_reply":"2022-11-13T18:42:23.117316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nt0start = time.time()\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport sys\n\nimport matplotlib.pyplot as plt\n#plt.style.use('dark_background')\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:42:23.120273Z","iopub.execute_input":"2022-11-13T18:42:23.121118Z","iopub.status.idle":"2022-11-13T18:42:23.126806Z","shell.execute_reply.started":"2022-11-13T18:42:23.121080Z","shell.execute_reply":"2022-11-13T18:42:23.125631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_meta = pd.read_csv('/kaggle/input/data-for-multimodal-singlecell-integration/_citeseq_meta_all_text_also.csv', index_col = 0)\ndf_meta_full = df_meta.copy()\ndf_meta = df_meta.iloc[:70988,:]\ndf_meta","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:42:23.128181Z","iopub.execute_input":"2022-11-13T18:42:23.129443Z","iopub.status.idle":"2022-11-13T18:42:23.503331Z","shell.execute_reply.started":"2022-11-13T18:42:23.129407Z","shell.execute_reply":"2022-11-13T18:42:23.502062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_Y = pd.read_csv('/kaggle/input/data-for-multimodal-singlecell-integration/CITEseq_targets_rescaled.csv',index_col = 0)\nY_true = df_Y.values\nlist_all_targets = list(df_Y.columns)\nprint(list_all_targets[:10] )\nprint(Y_true.shape)\ndf_Y","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:42:23.505487Z","iopub.execute_input":"2022-11-13T18:42:23.505860Z","iopub.status.idle":"2022-11-13T18:42:27.277684Z","shell.execute_reply.started":"2022-11-13T18:42:23.505827Z","shell.execute_reply":"2022-11-13T18:42:27.276393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ho = pd.read_csv('/kaggle/input/data-for-multimodal-singlecell-integration/df_save_hold_out.csv')\nmask_holdout = df_ho['HoldOut'] == 1\nmask_main = df_ho['HoldOut'] == 0\nprint(mask_holdout.sum(), mask_main.sum(), mask_holdout.sum()+ mask_main.sum(),df_ho.shape   )\ndf_ho.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:42:27.279181Z","iopub.execute_input":"2022-11-13T18:42:27.279603Z","iopub.status.idle":"2022-11-13T18:42:27.368315Z","shell.execute_reply.started":"2022-11-13T18:42:27.279533Z","shell.execute_reply":"2022-11-13T18:42:27.367034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# /kaggle/input/xgb-predictions-for-multimodal-competition/XGB_NFeat2719Y_pred_submission_Kaggle_way.csv\n# /kaggle/input/xgb-predictions-for-multimodal-competition/XGB_NFeat2719Y_pred_oof_private_like.csv\n\n# /kaggle/input/091seed112/XGB_NFeat2719Y_pred_submission_Kaggle_way.csv\n# /kaggle/input/091seed112/XGB_NFeat2719Y_pred_oof_private_like.csv\n\n# /kaggle/input/xgb-1211/XGB_NFeat2719Y_pred_submission_Kaggle_way.csv\n# /kaggle/input/xgb-1211/XGB_NFeat2719Y_pred_oof_private_like.csv\n\n# /kaggle/input/data-multimodal-singlecell-integration/XGBseed446/XGB_NFeat2719Y_pred_submission_Kaggle_way.csv\n# /kaggle/input/data-multimodal-singlecell-integration/XGBseed446/XGB_NFeat2719Y_pred_oof_private_like.csv\n\n# /kaggle/input/data-multimodal-singlecell-integration/XGBseed112/XGB_NFeat2719Y_pred_submission_Kaggle_way.csv\n# /kaggle/input/data-multimodal-singlecell-integration/XGBseed112/XGB_NFeat2719Y_pred_oof_private_like.csv\n\n# /kaggle/input/data-multimodal-singlecell-integration/XGBseed444/XGB_NFeat2719Y_pred_submission_Kaggle_way.csv\n# /kaggle/input/data-multimodal-singlecell-integration/XGBseed444/XGB_NFeat2719Y_pred_oof_private_like.csv","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:42:27.370060Z","iopub.execute_input":"2022-11-13T18:42:27.370660Z","iopub.status.idle":"2022-11-13T18:42:27.377607Z","shell.execute_reply.started":"2022-11-13T18:42:27.370610Z","shell.execute_reply":"2022-11-13T18:42:27.376116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_f = [\n'/kaggle/input/data-multimodal-singlecell-integration/MLP2_NFeat651/MLP2_NFeat651Y_pred_oof_private_like.csv',\n'/kaggle/input/data-multimodal-singlecell-integration/1dcnn_oof_3/1dcnn_v10_2719Y_pred_oof_private_like.csv',\n'/kaggle/input/mlp-ver6-outputs/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/mlp-ver6-seed0/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/mlp-ver6-seed1/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/mlp-ver6-seed2/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/mlp-ver6-seed3/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/mlp-ver6-seed4/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/mlp-ver6-seed5/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/mlp-ver6-seed6/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/mlp-ver6-seed7/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/mlp-ver6-seed9/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/xgb-predictions-for-multimodal-competition/XGB_NFeat2719Y_pred_oof_private_like.csv',\n'/kaggle/input/091seed112/XGB_NFeat2719Y_pred_oof_private_like.csv',\n'/kaggle/input/xgb-1211/XGB_NFeat2719Y_pred_oof_private_like.csv',\n'/kaggle/input/data-multimodal-singlecell-integration/XGBseed446/XGB_NFeat2719Y_pred_oof_private_like.csv',\n'/kaggle/input/data-multimodal-singlecell-integration/XGBseed112/XGB_NFeat2719Y_pred_oof_private_like.csv',\n'/kaggle/input/data-multimodal-singlecell-integration/XGBseed444/XGB_NFeat2719Y_pred_oof_private_like.csv',\n'/kaggle/input/data-multimodal-singlecell-integration/XGBseed20/XGB_NFeat2291Y_pred_oof_private_like.csv',\n'/kaggle/input/data-multimodal-singlecell-integration/XGBseed21/XGB_NFeat2291Y_pred_oof_private_like.csv',\n'/kaggle/input/088-200xgb-nfeat2719y-pred-submission-kaggle-way/XGB_NFeat2719Y_pred_oof_private_like.csv',\n'/kaggle/input/087-300-xgb-nfeat2719y-pred-submission-kaggle-way/XGB_NFeat2719Y_pred_oof_private_like (1).csv',\n'/kaggle/input/data4-for-mmscel/XGB_NFeat2719Y_pred_oof_private_like.csv',\n'/kaggle/input/data3-for-mmscel/XGB_NFeat2719Y_pred_oof_private_like.csv',\n'/kaggle/input/data5-for-mmscel/XGB_NFeat2719Y_pred_oof_private_like.csv',\n'/kaggle/input/data-multimodal-singlecell-integration/RidgeNfeat2719_Alpha1e4/RidgeNfeat2719_Alpha1e4_Y_pred_oof_private_like.csv',\n'/kaggle/input/blend-results/n_blends = 10 train_size = 0.9/RidgeNfeat2719_Alpha1e4_Y_pred_oof_private_like.csv',\n'/kaggle/input/blend-results/n_blends = 10 train_size = 0.75/RidgeNfeat2719_Alpha1e4_Y_pred_oof_private_like.csv',\n'/kaggle/input/blend-results/n_blends = 10 train_size = 0.5/RidgeNfeat2719_Alpha1e4_Y_pred_oof_private_like.csv',\n'/kaggle/input/blend-results/n_blends = 10 train_size = 1/RidgeNfeat2719_Alpha1e4_Y_pred_oof_private_like.csv',   \n'/kaggle/input/nn-for-multimodal-singlecell-integration/ResnetS1/MLP_ver6_shevY_pred_oof_private_like.csv',\n#'/kaggle/input/nn-for-multimodal-singlecell-integration/ResnetS2/MLP_ver6_shevY_pred_oof_private_like.csv ',  \n'/kaggle/input/nn-for-multimodal-singlecell-integration/ResnetS3/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/nn2-multimodal-singlecell-integration/ResnetS4/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/nn2-multimodal-singlecell-integration/ResnetS5/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/resnet-seed-6/MLP_ver6_shevY_pred_oof_private_like.csv',\n'/kaggle/input/mmscel-cvmodeling-advanced-seeds-60-61-62/Seed_60/LGB_NFeat160Y_pred_oof_private_like.csv',\n'/kaggle/input/mmscel-cvmodeling-advanced-seeds-60-61-62/Seed_62/LGB_NFeat160Y_pred_oof_private_like.csv',\n'/kaggle/input/three-lgbms/rs71/LGB_NFeat160Y_pred_oof_private_like.csv',\n'/kaggle/input/three-lgbms/rs72/LGB_NFeat160Y_pred_oof_private_like.csv',\n'/kaggle/input/three-lgbms/rs70/LGB_NFeat160Y_pred_oof_private_like.csv',\n'/kaggle/input/blend-results/MLP n_blends = 10 train_size = 0.9/MLP2_NFeat651Y_pred_oof_private_like.csv',\n'/kaggle/input/blend-results/MLP n_blends = 10 train_size = 0.75/MLP2_NFeat651Y_pred_oof_private_like.csv',\n# '/kaggle/input/catboost-oofpredicts/CatboostY_pred_oof_private_like.csv',\n# '/kaggle/input/lgbm-oofpredictss/LGBMY_pred_oof_private_like.csv',\n# '/kaggle/input/krr-2000feat-140tar-05trainsize/KernelRidgeNfeat2000_Alpha0.2_RBF_length10Y_pred_oof_private_like.csv',\n# '/kaggle/input/krr-2000feat-140tar-03trainsize/KernelRidgeNfeat2000_Alpha0.2_RBF_length10Y_pred_oof_private_like (1).csv',\n]\n\nlist_names = ['MLP2','1DCNN',\n              'KerasMLPV6', 'KerasMLPV6_RS0','KerasMLPV6_RS1','KerasMLPV6_RS2','KerasMLPV6_RS3',\n              'KerasMLPV6_RS4', 'KerasMLPV6_RS5', 'KerasMLPV6_RS6', 'KerasMLPV6_RS7', 'KerasMLPV6_RS9',\n              'XGB1','XGB2','XGB3','XGB4','XGB5','XGB6','XGB7','XGB8','XGB9','XGB10','XGB11','XGB12','XGB13',\n              'Ridge1', 'Ridge2', 'Ridge3', 'Ridge4','Ridge5',\n              'ResnetS1','ResnetS2','ResnetS3','ResnetS4','ResnetS5',\n              #'ResnetS6',\n              'LGBweak1', 'LGBweak2', 'LGBweak3', 'LGBweak4', 'LGBweak5'  ,\n              'MLPnew1','MLPnew2',\n              #'CatBoost', 'LGB','KRR05',  'KRR03'\n             ]\n\nlist_groups = ['MLP2', '1DCNN', 'KerasMLP', 'XGB','Resnet', 'Ridge','LGBweak','MLPnew']\n\ndef get_model_group( str_model_inf):\n    str_model_group = 'unknown'\n    for gr in list_groups:\n        if gr in str_model_inf:\n            str_model_group = gr\n    return str_model_group\n    \n\nif 1:\n    vec_groups_multiplicities = np.zeros(len(list_groups ))\n    for nm in list_names:\n        for gr in list_groups:\n            if gr in nm:\n                vec_groups_multiplicities[list_groups.index(gr)]+=1\n    print(vec_groups_multiplicities)           \nif 2:\n    vec_groups_multiplicities = np.zeros(len(list_groups ))\n    for nm in list_names:\n        str_model_group = get_model_group( nm )\n        vec_groups_multiplicities[list_groups.index(str_model_group) ]+=1\n    print(vec_groups_multiplicities)            \n    \n\ndf_groups = pd.DataFrame(index = list_groups, data =vec_groups_multiplicities, columns = ['Group Size']  )\ndisplay(df_groups)\n\nprint(len(list_f), len(list_names))\n\nfrom os.path import exists\nfor i,f in enumerate(list_f):\n    if not exists(f): \n        print('\\n\\n', 'NOT exists: ', f, '\\n\\n')\n    else:\n        print(list_names[i],f )\n        ","metadata":{"execution":{"iopub.status.busy":"2022-11-13T19:52:45.009635Z","iopub.execute_input":"2022-11-13T19:52:45.010073Z","iopub.status.idle":"2022-11-13T19:52:45.110480Z","shell.execute_reply.started":"2022-11-13T19:52:45.010039Z","shell.execute_reply":"2022-11-13T19:52:45.109265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nrescale_predicts_to_mean0_std1 = True\nn_models = len(list_f)\n\ndict_df = {}\nfor i,f in enumerate(list_f[:n_models]):\n    key_loc = list_names[i]\n    df = pd.read_csv(f,index_col = 0)\n    if rescale_predicts_to_mean0_std1:\n        t = df.values\n        t -= t.mean(axis=1).reshape(-1, 1)\n        t /= t.std(axis=1).reshape(-1, 1)\n        df = pd.DataFrame(t, index = df.index, columns  = df.columns )\n    print(df.shape)\n    display(df.head(2))\n    dict_df[key_loc] = df","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:42:27.443780Z","iopub.execute_input":"2022-11-13T18:42:27.444494Z","iopub.status.idle":"2022-11-13T18:44:20.916736Z","shell.execute_reply.started":"2022-11-13T18:42:27.444441Z","shell.execute_reply":"2022-11-13T18:44:20.915091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# '/kaggle/input/lgbm-oofpredictss/LGBMY_pred_oof_private_like.csv',\n# /kaggle/input/lgbm-oofpredictss/LGBMY_pred_submission_Kaggle_way.csv\n\ndict_df_submit = {}\nfor i,f in enumerate(list_f[:n_models]):\n    key_loc = list_names[i]\n    f2 = f.replace('oof_private_like','submission_Kaggle_way' )\n    df = pd.read_csv(f2,index_col = 0)\n    if rescale_predicts_to_mean0_std1:\n        t = df.values\n        t -= t.mean(axis=1).reshape(-1, 1)\n        t /= t.std(axis=1).reshape(-1, 1)\n        df = pd.DataFrame(t, index = df.index, columns  = df.columns )\n    print(df.shape)\n    display(df.head(2))\n    dict_df_submit[key_loc] = df","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:44:20.918689Z","iopub.execute_input":"2022-11-13T18:44:20.919252Z","iopub.status.idle":"2022-11-13T18:45:35.139683Z","shell.execute_reply.started":"2022-11-13T18:44:20.919182Z","shell.execute_reply":"2022-11-13T18:45:35.138518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlation Scoring Function","metadata":{}},{"cell_type":"code","source":"import gc\nrescale_Y_to_mean0_std1 = True\n\ndef correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    \n    # Input should be matrices - does not make sense for vectors \n    if  len(y_pred.shape)< 2: return -10 # Some result to inform for incorrect input\n    if  y_pred.shape[1] < 2: return -10 # Some result to inform for incorrect input\n\n    y2 = y_pred.copy()\n    y2 -= y2.mean(axis=1).reshape(-1, 1);    y2 /= y2.std(axis=1).reshape(-1, 1)    \n    if rescale_Y_to_mean0_std1:\n        y1 = y_true # Already rescaled \n    else:\n        y1 = y_true.copy(); \n        y1 -= y1.mean(axis=1).reshape(-1, 1);    y1 /= y1.std(axis=1).reshape(-1, 1) \n        \n    c = (y1*y2).mean().mean()# Correlation for rescaled matrices is just matrix product and average \n    \n    c = (y1*y2).mean().mean()# Correlation for rescaled matrices is just matrix product and average \n    \n    # Memory control:\n    if not rescale_Y_to_mean0_std1:\n        del y1\n    del y2\n    gc.collect()\n    \n    return c\n\n    # Slow way:\n    \n    if type(y_true) == pd.DataFrame: y_true = y_true.values\n    if type(y_pred) == pd.DataFrame: y_pred = y_pred.values\n    corrsum = 0\n    for i in range(len(y_true)):\n        corrsum += np.corrcoef(y_true[i], y_pred[i])[1, 0]\n    return corrsum / len(y_true)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:45:35.140939Z","iopub.execute_input":"2022-11-13T18:45:35.142160Z","iopub.status.idle":"2022-11-13T18:45:35.155401Z","shell.execute_reply.started":"2022-11-13T18:45:35.142113Z","shell.execute_reply":"2022-11-13T18:45:35.153941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlation scores before blend","metadata":{"execution":{"iopub.status.busy":"2022-11-11T17:25:31.524570Z","iopub.execute_input":"2022-11-11T17:25:31.525081Z","iopub.status.idle":"2022-11-11T17:25:31.530934Z","shell.execute_reply.started":"2022-11-11T17:25:31.525042Z","shell.execute_reply":"2022-11-11T17:25:31.529366Z"}}},{"cell_type":"code","source":"%%time\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import r2_score\n\ndf_stat = pd.DataFrame()\nIX = 0\nfor i,key_loc in enumerate(list( dict_df.keys())  ):\n    key_loc = list_names[i]\n    df = dict_df[key_loc]\n    print(key_loc, df.shape)\n    #display(df.head(2))\n    s = correlation_score(Y_true,df.values)\n    df_stat.loc[IX,'Model'] = key_loc\n    df_stat.loc[IX,'Corr'] = s\n    df_stat.loc[IX,'r2'] = r2_score(Y_true,df.values)\n    df_stat.loc[IX,'MSE'] = mean_squared_error(Y_true,df.values)\n    \n    m = mask_holdout; postfix = ' HO'\n    df_stat.loc[IX,'Corr'+ postfix] = correlation_score(Y_true[m],df[m].values)\n    df_stat.loc[IX,'r2'+ postfix] = r2_score(Y_true[m],df[m].values)\n    df_stat.loc[IX,'MSE'+ postfix] = mean_squared_error(Y_true[m],df[m].values)\n    m = mask_main;  postfix = ' Main'\n    df_stat.loc[IX,'Corr'+postfix] = correlation_score(Y_true[m],df[m].values)\n    df_stat.loc[IX,'r2'+postfix] = r2_score(Y_true[m],df[m].values)\n    df_stat.loc[IX,'MSE'+postfix] = mean_squared_error(Y_true[m],df[m].values)\n    \n    IX += 1\n    \ndisplay(df_stat)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:45:35.157364Z","iopub.execute_input":"2022-11-13T18:45:35.157866Z","iopub.status.idle":"2022-11-13T18:46:44.000163Z","shell.execute_reply.started":"2022-11-13T18:45:35.157814Z","shell.execute_reply":"2022-11-13T18:46:43.998490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndd = df_stat.sort_values('Corr', ascending = False)\nprint(dd['Model'][dd['Corr']>0.8923].values)\nprint(dd['Model'].iloc[5:10].values)\n\ndd","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:46:44.002570Z","iopub.execute_input":"2022-11-13T18:46:44.003081Z","iopub.status.idle":"2022-11-13T18:46:44.043485Z","shell.execute_reply.started":"2022-11-13T18:46:44.003030Z","shell.execute_reply":"2022-11-13T18:46:44.042076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dd.columns","metadata":{"execution":{"iopub.status.busy":"2022-11-13T19:39:20.241259Z","iopub.execute_input":"2022-11-13T19:39:20.242689Z","iopub.status.idle":"2022-11-13T19:39:20.251713Z","shell.execute_reply.started":"2022-11-13T19:39:20.242621Z","shell.execute_reply":"2022-11-13T19:39:20.250474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dd['Group'] = dd['Model'].apply( get_model_group )\ntmp = dd.groupby('Group')[ 'Corr'].mean()\ntmp.name = 'Mean Corr'\ndf_groups2 = df_groups.join(tmp)\n\ntmp = dd.groupby('Group')[ 'Corr'].max()\ntmp.name = 'Max Corr'\ndf_groups2 = df_groups2.join(tmp)\n\ntmp = dd.groupby('Group')[ 'Corr'].min()\ntmp.name = 'Min Corr'\ndf_groups2 = df_groups2.join(tmp)\n\ndf_groups2 = df_groups2.sort_values('Mean Corr', ascending = False)\ndf_groups2","metadata":{"execution":{"iopub.status.busy":"2022-11-13T20:52:34.894543Z","iopub.execute_input":"2022-11-13T20:52:34.894995Z","iopub.status.idle":"2022-11-13T20:52:34.926024Z","shell.execute_reply.started":"2022-11-13T20:52:34.894956Z","shell.execute_reply":"2022-11-13T20:52:34.924683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Blend","metadata":{}},{"cell_type":"code","source":"df_meta['day'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:46:44.044977Z","iopub.execute_input":"2022-11-13T18:46:44.045410Z","iopub.status.idle":"2022-11-13T18:46:44.057662Z","shell.execute_reply.started":"2022-11-13T18:46:44.045372Z","shell.execute_reply":"2022-11-13T18:46:44.056503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n#sklearn.linear_model.LinearRegression\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import Lasso\nreg = LinearRegression()\n#reg = Lasso(alpha = 0.01)\nreg = Lasso(alpha = .04, positive = True , fit_intercept=False)\n\nlist_keys4blend = list( dict_df.keys())\nprint(list_keys4blend )\n\nmask_day = (df_meta['day']!=4).values\nmask4blend = (mask_day) & (mask_main) # mask_main - excludes Holdout\nprint(mask4blend.sum())\n\ndf_blend = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\ndf_tmp = pd.DataFrame()\nfor i_col, col in enumerate(list_all_targets):\n    for i,key_loc in enumerate( list_keys4blend  ):\n        df_tmp[key_loc] = dict_df[key_loc][col]\n    \n    m = mask4blend\n    reg.fit(df_tmp.values[m], Y_true[m,i_col]); \n    df_blend[col] = reg.predict(df_tmp.values)\n    #print(i_col,  col ,reg.coef_,  r2_score(Y_true[:,i_col], df_blend[col] ),  mean_squared_error(Y_true[:,i_col], df_blend[col] ),  )\n\ndf_blend    ","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:46:44.059452Z","iopub.execute_input":"2022-11-13T18:46:44.060287Z","iopub.status.idle":"2022-11-13T18:48:03.720459Z","shell.execute_reply.started":"2022-11-13T18:46:44.060243Z","shell.execute_reply":"2022-11-13T18:48:03.719087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Blend by average","metadata":{}},{"cell_type":"code","source":"df_blend_just_average = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\nfor i,key_loc in enumerate( list_keys4blend  ):\n    df_blend_just_average += dict_df[key_loc].values\ndf_blend_just_average /= (i+1)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:03.722189Z","iopub.execute_input":"2022-11-13T18:48:03.722632Z","iopub.status.idle":"2022-11-13T18:48:04.668143Z","shell.execute_reply.started":"2022-11-13T18:48:03.722593Z","shell.execute_reply":"2022-11-13T18:48:04.666755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Blend by average just KERAS MLP","metadata":{}},{"cell_type":"code","source":"print(list_keys4blend)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:04.669918Z","iopub.execute_input":"2022-11-13T18:48:04.670317Z","iopub.status.idle":"2022-11-13T18:48:04.678292Z","shell.execute_reply.started":"2022-11-13T18:48:04.670282Z","shell.execute_reply":"2022-11-13T18:48:04.676428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_blend_just_average_KERAS = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\nc = 0\nfor i,key_loc in enumerate( list_keys4blend  ):\n    if 'Keras'in key_loc:\n        df_blend_just_average_KERAS += dict_df[key_loc].values; c+=1\ndf_blend_just_average_KERAS /= (c)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:04.680676Z","iopub.execute_input":"2022-11-13T18:48:04.681063Z","iopub.status.idle":"2022-11-13T18:48:04.958581Z","shell.execute_reply.started":"2022-11-13T18:48:04.681030Z","shell.execute_reply":"2022-11-13T18:48:04.957049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Blend KERAS top5","metadata":{}},{"cell_type":"code","source":"ll_loc = ['KerasMLPV6_RS1', 'KerasMLPV6_RS3', 'KerasMLPV6_RS0', 'KerasMLPV6_RS2', 'KerasMLPV6_RS7']\ndf_blend_just_average_KERAS_top5 = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\nc = 0\nfor i,key_loc in enumerate( list_keys4blend  ):\n    if key_loc in ll_loc:\n        df_blend_just_average_KERAS_top5 += dict_df[key_loc].values; c+=1\ndf_blend_just_average_KERAS_top5 /= (c)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:04.959961Z","iopub.execute_input":"2022-11-13T18:48:04.960447Z","iopub.status.idle":"2022-11-13T18:48:05.124328Z","shell.execute_reply.started":"2022-11-13T18:48:04.960408Z","shell.execute_reply":"2022-11-13T18:48:05.122623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ll_loc = ['KerasMLPV6_RS5', 'KerasMLPV6_RS9', 'KerasMLPV6_RS6', 'KerasMLPV6_RS4',\n 'KerasMLPV6']\ndf_blend_just_average_KERAS_bottom5 = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\nc = 0\nfor i,key_loc in enumerate( list_keys4blend  ):\n    if key_loc in ll_loc:\n        df_blend_just_average_KERAS_bottom5 += dict_df[key_loc].values; c+=1\ndf_blend_just_average_KERAS_bottom5 /= (c)\ndf_blend_just_average_KERAS_bottom5.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:05.126356Z","iopub.execute_input":"2022-11-13T18:48:05.126861Z","iopub.status.idle":"2022-11-13T18:48:05.335240Z","shell.execute_reply.started":"2022-11-13T18:48:05.126807Z","shell.execute_reply":"2022-11-13T18:48:05.334300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost blend","metadata":{}},{"cell_type":"code","source":"df_blend_just_average_XGB = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\nc = 0\nfor i,key_loc in enumerate( list_keys4blend  ):\n    if 'XGB' in key_loc:\n        df_blend_just_average_XGB += dict_df[key_loc].values; c+=1\ndf_blend_just_average_XGB /= (c)\ndisplay(df_blend_just_average_XGB.head(1) )\n\ndf_blend_just_average_XGB_first5 = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\nc = 0\nfor i,key_loc in enumerate( list_keys4blend  ):\n    if key_loc in ['XGB1', 'XGB2', 'XGB3', 'XGB4', 'XGB5', ]:\n        df_blend_just_average_XGB_first5 += dict_df[key_loc].values; c+=1\ndf_blend_just_average_XGB_first5 /= (c)\ndisplay(df_blend_just_average_XGB_first5.head(1) )","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:05.341377Z","iopub.execute_input":"2022-11-13T18:48:05.342916Z","iopub.status.idle":"2022-11-13T18:48:05.852373Z","shell.execute_reply.started":"2022-11-13T18:48:05.342844Z","shell.execute_reply":"2022-11-13T18:48:05.850715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LGBWeak","metadata":{}},{"cell_type":"code","source":"df_blend_just_average_LGBweak = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\nc = 0\nfor i,key_loc in enumerate( list_keys4blend  ):\n    if 'LGBweak' in key_loc:\n        df_blend_just_average_LGBweak += dict_df[key_loc].values; c+=1\ndf_blend_just_average_LGBweak /= (c)\ndisplay(df_blend_just_average_LGBweak.head(1) )","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:05.854337Z","iopub.execute_input":"2022-11-13T18:48:05.854935Z","iopub.status.idle":"2022-11-13T18:48:06.034188Z","shell.execute_reply.started":"2022-11-13T18:48:05.854882Z","shell.execute_reply":"2022-11-13T18:48:06.032910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Resnet","metadata":{}},{"cell_type":"code","source":"df_blend_just_average_Resnet = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\nc = 0\nfor i,key_loc in enumerate( list_keys4blend  ):\n    if 'Resnet' in key_loc:\n        df_blend_just_average_Resnet += dict_df[key_loc].values; c+=1\ndf_blend_just_average_Resnet /= (c)\ndisplay(df_blend_just_average_Resnet.head(1) )","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:06.035724Z","iopub.execute_input":"2022-11-13T18:48:06.036236Z","iopub.status.idle":"2022-11-13T18:48:06.214921Z","shell.execute_reply.started":"2022-11-13T18:48:06.036162Z","shell.execute_reply":"2022-11-13T18:48:06.213789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ridge","metadata":{}},{"cell_type":"code","source":"df_blend_just_average_Ridge = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\nc = 0\nfor i,key_loc in enumerate( list_keys4blend  ):\n    if 'Ridge' in key_loc:\n        df_blend_just_average_Ridge += dict_df[key_loc].values; c+=1\ndf_blend_just_average_Ridge /= (c)\ndisplay(df_blend_just_average_Ridge.head(1) )","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:06.218895Z","iopub.execute_input":"2022-11-13T18:48:06.219305Z","iopub.status.idle":"2022-11-13T18:48:06.397630Z","shell.execute_reply.started":"2022-11-13T18:48:06.219269Z","shell.execute_reply":"2022-11-13T18:48:06.396120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## MLPnew","metadata":{}},{"cell_type":"code","source":"df_blend_just_average_MLPnew = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\nc = 0\nfor i,key_loc in enumerate( list_keys4blend  ):\n    if 'MLPnew' in key_loc:\n        df_blend_just_average_MLPnew += dict_df[key_loc].values; c+=1\ndf_blend_just_average_MLPnew /= (c)\ndisplay(df_blend_just_average_MLPnew.head(1) )","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:06.399248Z","iopub.execute_input":"2022-11-13T18:48:06.399684Z","iopub.status.idle":"2022-11-13T18:48:06.510693Z","shell.execute_reply.started":"2022-11-13T18:48:06.399644Z","shell.execute_reply":"2022-11-13T18:48:06.509272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Scores for blend and  solo models ","metadata":{}},{"cell_type":"code","source":"%%time\ndf_stat_blend = pd.DataFrame()\nIX = 0\nll_loc = ['Blend All','Blend Keras','Blend Keras Top5','Blend Keras Bottom5', 'XGB', 'XGBfirst5',\n          'LGBweak', 'Resnet', 'Ridge',\n          'Blend Lasso']\nfor i,key_loc in enumerate( ll_loc + list( dict_df.keys())  ):\n    if key_loc == 'Blend All':\n        df_loc = df_blend_just_average\n        nm_loc = 'Blend All Average'\n    elif key_loc == 'Blend Keras':\n        df_loc = df_blend_just_average_KERAS\n        nm_loc = 'Blend Keras'\n    elif key_loc == 'Blend Keras Top5':\n        df_loc = df_blend_just_average_KERAS_top5\n        nm_loc = 'Blend Keras Top5'\n    elif key_loc == 'Blend Keras Bottom5':\n        df_loc = df_blend_just_average_KERAS_bottom5\n        nm_loc = 'Blend Keras Bottom5'\n    elif key_loc == 'XGB':\n        df_loc = df_blend_just_average_XGB\n        nm_loc = 'Blend XGB All'\n    elif key_loc == 'XGBfirst5':\n        df_loc = df_blend_just_average_XGB_first5\n        nm_loc = 'Blend XGBfirst5'\n    elif key_loc == 'LGBweak':\n        df_loc = df_blend_just_average_LGBweak\n        nm_loc = 'Blend LGBweak'\n    elif key_loc == 'Ridge':\n        df_loc = df_blend_just_average_Ridge\n        nm_loc = 'Blend Ridge'\n    elif key_loc == 'Resnet':\n        df_loc = df_blend_just_average_Resnet\n        nm_loc = 'Blend Resnet'\n    elif key_loc == 'Blend Lasso':\n        df_loc = df_blend\n        nm_loc = 'Blend Lasso'\n    else:\n        #key_loc = list_names[i]\n        df_loc = dict_df[key_loc]\n        nm_loc = key_loc\n        \n    df_stat_blend.loc[IX,'Model'] = nm_loc\n    \n    \n    print(key_loc, nm_loc, df_loc.shape)\n    m = np.ones_like(Y_true[:,0], dtype = bool);  postfix = ' ALL'\n    df_stat_blend.loc[IX,'Corr'+postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    m = mask_holdout; postfix = ' HO'\n    df_stat_blend.loc[IX,'Corr'+ postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    \n    \n    m = np.ones_like(Y_true[:,0], dtype = bool);  postfix = ' ALL'\n    df_stat_blend.loc[IX,'Corr'+postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    df_stat_blend.loc[IX,'r2'+postfix] = r2_score(Y_true[m],df_loc.values[m])\n    df_stat_blend.loc[IX,'MSE'+postfix] = mean_squared_error(Y_true[m],df_loc.values[m])\n    \n    m = mask_holdout; postfix = ' HO'\n    df_stat_blend.loc[IX,'Corr'+ postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    df_stat_blend.loc[IX,'r2'+ postfix] = r2_score(Y_true[m],df_loc.values[m])\n    df_stat_blend.loc[IX,'MSE'+ postfix] = mean_squared_error(Y_true[m],df_loc.values[m])\n    \n    m = (~mask4blend); postfix = ' Tst'\n    df_stat_blend.loc[IX,'Corr'+ postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    df_stat_blend.loc[IX,'r2'+ postfix] = r2_score(Y_true[m],df_loc.values[m])\n    df_stat_blend.loc[IX,'MSE'+ postfix] = mean_squared_error(Y_true[m],df_loc.values[m])\n    \n    m = (mask4blend); postfix = ' Tr'\n    df_stat_blend.loc[IX,'Corr'+ postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    df_stat_blend.loc[IX,'r2'+ postfix] = r2_score(Y_true[m],df_loc.values[m])\n    df_stat_blend.loc[IX,'MSE'+ postfix] = mean_squared_error(Y_true[m],df_loc.values[m])\n    \n#     df_stat_blend.loc[IX,'Corr'] = correlation_score(Y_true,df_loc.values)\n#     df_stat_blend.loc[IX,'r2'] = r2_score(Y_true,df_loc.values)\n#     df_stat_blend.loc[IX,'MSE'] = mean_squared_error(Y_true,df_loc.values)\n\n    \n    m = mask_main;  postfix = ' Main'\n    df_stat_blend.loc[IX,'Corr'+postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    df_stat_blend.loc[IX,'r2'+postfix] = r2_score(Y_true[m],df_loc.values[m])\n    df_stat_blend.loc[IX,'MSE'+postfix] = mean_squared_error(Y_true[m],df_loc.values[m])\n\n    \n    IX += 1\ndf_stat_blend    ","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:48:06.512318Z","iopub.execute_input":"2022-11-13T18:48:06.512825Z","iopub.status.idle":"2022-11-13T18:50:57.648945Z","shell.execute_reply.started":"2022-11-13T18:48:06.512782Z","shell.execute_reply":"2022-11-13T18:50:57.647407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_stat_blend.sort_values('Corr ALL',ascending = False)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:50:57.650898Z","iopub.execute_input":"2022-11-13T18:50:57.651338Z","iopub.status.idle":"2022-11-13T18:50:57.699512Z","shell.execute_reply.started":"2022-11-13T18:50:57.651300Z","shell.execute_reply":"2022-11-13T18:50:57.698118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weighted Blend Experiments ","metadata":{}},{"cell_type":"code","source":"df_groups3 = df_groups2.copy()\ndf_groups3","metadata":{"execution":{"iopub.status.busy":"2022-11-13T21:07:40.042940Z","iopub.execute_input":"2022-11-13T21:07:40.043413Z","iopub.status.idle":"2022-11-13T21:07:40.059541Z","shell.execute_reply.started":"2022-11-13T21:07:40.043375Z","shell.execute_reply":"2022-11-13T21:07:40.058286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_groups3.shape)\ndf_groups3['Ranks Top4'] = [1,1,1,1,0,0,0,0]\ndf_groups3['Ranks All'] = [1,1,1,1,1,1,1,1]\ndf_groups3['Ranks Manual'] = [2,2,2,1,0.2,0.5,0.1,0.1]\n\ndict_ranks_tmp = {'MLPnew': 1, 'Resnet':1 , 'KerasMLP:':1, 'XGB':0, 'MLP2':0, 'Ridge':0, 'LGBweak':0,'1DCNN':0}\ndf_groups3['Ranks Top3'] = [dict_ranks_tmp[k] for k in dict_ranks_tmp]\ndf_groups3['Ranks Only MLPnew'] = [1,0,0,0,0,0,0,0]\ndf_groups3['Ranks Only Resnet'] = [0,1,0,0,0,0,0,0]\ndf_groups3['Ranks Top5'] = [1,1,1,1,1,0,0,0]\ndf_groups3['Ranks Top6'] = [1,1,1,1,1,1,0,0]\ndf_groups3['Ranks Top6 +bit'] = [1,1,1,1,1,1,0.2,0.2]\n\nfor col in df_groups3.columns:\n    if not 'Ranks' in col: continue\n    nm = 'Weights' + col[5:]\n    df_groups3[nm] = df_groups3[col]/df_groups3[col].sum() \n    df_groups3[nm] = df_groups3[nm]/df_groups3['Group Size']\ndf_groups3","metadata":{"execution":{"iopub.status.busy":"2022-11-13T21:07:41.144312Z","iopub.execute_input":"2022-11-13T21:07:41.144725Z","iopub.status.idle":"2022-11-13T21:07:41.201925Z","shell.execute_reply.started":"2022-11-13T21:07:41.144692Z","shell.execute_reply":"2022-11-13T21:07:41.200679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict_blend_results = {}\nfor col in df_groups3.columns:\n    if not 'Weight' in col: continue\n    print(col)\n    df_blend_new = pd.DataFrame(np.zeros(Y_true.shape), columns = list_all_targets, index = df_Y.index)\n    for i,key_loc in enumerate( list_keys4blend  ):\n        model_group = get_model_group(key_loc)\n        model_weight = df_groups3.loc[model_group, col]\n        df_blend_new += model_weight*dict_df[key_loc].values#; c+=1\n\n    dict_blend_results[col] =  df_blend_new\ndisplay(dict_blend_results[col].head(3)) ","metadata":{"execution":{"iopub.status.busy":"2022-11-13T21:07:44.431844Z","iopub.execute_input":"2022-11-13T21:07:44.432703Z","iopub.status.idle":"2022-11-13T21:08:03.260817Z","shell.execute_reply.started":"2022-11-13T21:07:44.432665Z","shell.execute_reply":"2022-11-13T21:08:03.259287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_stat_blend_weighted  = pd.DataFrame()\nIX = 0 \nfor i,key_loc in enumerate( dict_blend_results.keys()  ):\n    \n    nm_loc = key_loc\n    df_stat_blend.loc[IX,'Model'] = nm_loc\n    \n    df_loc = dict_blend_results[key_loc]\n        \n    df_stat_blend_weighted.loc[IX,'Model'] = nm_loc\n    \n    \n    print(key_loc, nm_loc, df_loc.shape)\n    m = np.ones_like(Y_true[:,0], dtype = bool);  postfix = ' ALL'\n    df_stat_blend_weighted.loc[IX,'Corr'+postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    m = mask_holdout; postfix = ' HO'\n    df_stat_blend_weighted.loc[IX,'Corr'+ postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    \n    \n    m = np.ones_like(Y_true[:,0], dtype = bool);  postfix = ' ALL'\n    df_stat_blend_weighted.loc[IX,'Corr'+postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    df_stat_blend_weighted.loc[IX,'r2'+postfix] = r2_score(Y_true[m],df_loc.values[m])\n    df_stat_blend_weighted.loc[IX,'MSE'+postfix] = mean_squared_error(Y_true[m],df_loc.values[m])\n    \n    m = mask_holdout; postfix = ' HO'\n    df_stat_blend_weighted.loc[IX,'Corr'+ postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    df_stat_blend_weighted.loc[IX,'r2'+ postfix] = r2_score(Y_true[m],df_loc.values[m])\n    df_stat_blend_weighted.loc[IX,'MSE'+ postfix] = mean_squared_error(Y_true[m],df_loc.values[m])\n    \n    m = (~mask4blend); postfix = ' Tst'\n    df_stat_blend_weighted.loc[IX,'Corr'+ postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    df_stat_blend_weighted.loc[IX,'r2'+ postfix] = r2_score(Y_true[m],df_loc.values[m])\n    df_stat_blend_weighted.loc[IX,'MSE'+ postfix] = mean_squared_error(Y_true[m],df_loc.values[m])\n    \n    m = (mask4blend); postfix = ' Tr'\n    df_stat_blend_weighted.loc[IX,'Corr'+ postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    df_stat_blend_weighted.loc[IX,'r2'+ postfix] = r2_score(Y_true[m],df_loc.values[m])\n    df_stat_blend_weighted.loc[IX,'MSE'+ postfix] = mean_squared_error(Y_true[m],df_loc.values[m])\n    \n#     df_stat_blend.loc[IX,'Corr'] = correlation_score(Y_true,df_loc.values)\n#     df_stat_blend.loc[IX,'r2'] = r2_score(Y_true,df_loc.values)\n#     df_stat_blend.loc[IX,'MSE'] = mean_squared_error(Y_true,df_loc.values)\n\n    \n    m = mask_main;  postfix = ' Main'\n    df_stat_blend_weighted.loc[IX,'Corr'+postfix] = correlation_score(Y_true[m],df_loc.values[m])\n    df_stat_blend_weighted.loc[IX,'r2'+postfix] = r2_score(Y_true[m],df_loc.values[m])\n    df_stat_blend_weighted.loc[IX,'MSE'+postfix] = mean_squared_error(Y_true[m],df_loc.values[m])\n\n    \n    IX += 1\ndf_stat_blend_weighted\n","metadata":{"execution":{"iopub.status.busy":"2022-11-13T21:08:03.265026Z","iopub.execute_input":"2022-11-13T21:08:03.265420Z","iopub.status.idle":"2022-11-13T21:08:49.023369Z","shell.execute_reply.started":"2022-11-13T21:08:03.265386Z","shell.execute_reply":"2022-11-13T21:08:49.022029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare data for submission","metadata":{}},{"cell_type":"code","source":"dict_blend_results_submit = {}\nfor col in df_groups3.columns:\n    if not 'Weight' in col: continue\n    print(col)\n    df_blend_new = pd.DataFrame( np.zeros( (48663,140)),index= df_meta_full.index[70988:] , columns = list_all_targets, )\n    for i,key_loc in enumerate( list_keys4blend  ):\n        model_group = get_model_group(key_loc)\n        model_weight = df_groups3.loc[model_group, col]\n        df_blend_new += model_weight*dict_df_submit[key_loc].values#; c+=1\n\n    dict_blend_results_submit[col] =  df_blend_new\ndisplay(dict_blend_results_submit[col].head(3)) ","metadata":{"execution":{"iopub.status.busy":"2022-11-13T22:00:35.105433Z","iopub.execute_input":"2022-11-13T22:00:35.105944Z","iopub.status.idle":"2022-11-13T22:00:51.834585Z","shell.execute_reply.started":"2022-11-13T22:00:35.105902Z","shell.execute_reply":"2022-11-13T22:00:51.833170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_blend_just_average_submit = pd.DataFrame( np.zeros( (48663,140)) , \n        index= df_meta_full.index[70988:] , columns = list_all_targets, )\nfor i,key_loc in enumerate( list_keys4blend  ):\n    df_blend_just_average_submit += dict_df_submit[key_loc].values\ndf_blend_just_average_submit /= (i+1)\ndf_blend_just_average_submit","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:50:57.701452Z","iopub.execute_input":"2022-11-13T18:50:57.701886Z","iopub.status.idle":"2022-11-13T18:50:58.434298Z","shell.execute_reply.started":"2022-11-13T18:50:57.701850Z","shell.execute_reply":"2022-11-13T18:50:58.433023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_blend_save = dict_blend_results_submit['Weights Top4']# .keys()#[key_loc]\ndf_blend_save.to_csv('submit_averageTop4GroupsModels.csv' )\ndf_blend_save","metadata":{"execution":{"iopub.status.busy":"2022-11-13T22:01:26.953710Z","iopub.execute_input":"2022-11-13T22:01:26.954129Z","iopub.status.idle":"2022-11-13T22:01:37.627586Z","shell.execute_reply.started":"2022-11-13T22:01:26.954091Z","shell.execute_reply":"2022-11-13T22:01:37.626078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ndf_blend_just_average_submit.to_csv('submit_average4models.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:50:58.436338Z","iopub.execute_input":"2022-11-13T18:50:58.436721Z","iopub.status.idle":"2022-11-13T18:51:08.692938Z","shell.execute_reply.started":"2022-11-13T18:50:58.436686Z","shell.execute_reply":"2022-11-13T18:51:08.691249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('%.1f seconds passed total '%(time.time()-t0start) )","metadata":{"execution":{"iopub.status.busy":"2022-11-13T18:51:08.695429Z","iopub.execute_input":"2022-11-13T18:51:08.695846Z","iopub.status.idle":"2022-11-13T18:51:08.703862Z","shell.execute_reply.started":"2022-11-13T18:51:08.695808Z","shell.execute_reply":"2022-11-13T18:51:08.702879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}