{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport glob\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T06:22:32.979605Z","iopub.execute_input":"2022-08-10T06:22:32.980115Z","iopub.status.idle":"2022-08-10T06:22:33.984084Z","shell.execute_reply.started":"2022-08-10T06:22:32.980022Z","shell.execute_reply":"2022-08-10T06:22:33.980014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Blend Boosting study American Express - Default Prediction\nHere I share with you a systematic blend boosting study on dataset of the American Express - Default. I just collect some submission files (16 distincts, but skip some of them).\n\nBasically, I start to analysis of correlations, then decide to sort them according to their sum of correlation values in between. This lets me divide 16 scores into 3 subgroups. Then I make internal linear calibration in each subgroup by considering their scores on the Kaggle. Finally I make recalling between subgroups to achieve higher scores on the Kaggle by resubmission.\n\nThanks for@Hikmet Sezen","metadata":{}},{"cell_type":"code","source":"def amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:22:33.987407Z","iopub.execute_input":"2022-08-10T06:22:33.988510Z","iopub.status.idle":"2022-08-10T06:22:34.003137Z","shell.execute_reply.started":"2022-08-10T06:22:33.988447Z","shell.execute_reply":"2022-08-10T06:22:34.000848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading data including 16 best scores\ndf_sub = pd.read_csv(r'../input/blend-second/test_lgbm_baseline_5fold_seed_blend.csv')\n\n# a rough correlation based visualization of 16 best scores\nplt.figure(figsize=(10,10))\nsns.heatmap(df_sub.corr(), cmap='Spectral')\nplt.ylabel('file index numbers')\nplt.xlabel('file index numbers')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:22:34.005153Z","iopub.execute_input":"2022-08-10T06:22:34.005976Z","iopub.status.idle":"2022-08-10T06:22:40.796558Z","shell.execute_reply.started":"2022-08-10T06:22:34.005929Z","shell.execute_reply":"2022-08-10T06:22:40.795345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\ndf_mean_corr = pd.DataFrame({'mean_corr': df_sub.corr().mean()})\ndf_mean_corr = df_mean_corr.sort_values('mean_corr', ascending=False)\ndf_mean_corr = df_mean_corr.reset_index()\n\nplt.plot(df_mean_corr.index[:8], df_mean_corr['mean_corr'].values[:8], 'o', ms=10)\nplt.plot(df_mean_corr.index[8:13], df_mean_corr['mean_corr'].values[8:13], 'o', ms=10)\nplt.plot(df_mean_corr.index[13:14], df_mean_corr['mean_corr'].values[13:14], 'o', ms=10)\nplt.plot(df_mean_corr.index[14:15], df_mean_corr['mean_corr'].values[14:15], 'o', ms=10)\nplt.plot(df_mean_corr.index[15:16], df_mean_corr['mean_corr'].values[15:16], 'o', ms=10)\n\nplt.xticks([*range(len(df_mean_corr))], df_mean_corr['index'].tolist())\nplt.title('determination of sub_groups')\nplt.ylabel('a correlation ralated index')\nplt.xlabel('file index numbers')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:22:40.799747Z","iopub.execute_input":"2022-08-10T06:22:40.800230Z","iopub.status.idle":"2022-08-10T06:22:41.900029Z","shell.execute_reply.started":"2022-08-10T06:22:40.800184Z","shell.execute_reply":"2022-08-10T06:22:41.899177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub['weighted_avg'] = abs(1 * (\n    100 * ( 100 * df_sub['0'] + 40 * df_sub['1'] + 30 * df_sub['2'] + 20 * df_sub['5'] +\n           20 * df_sub['6'] + 10 * df_sub['9'] + 10 * df_sub['12'] + 10 * df_sub['16']) / 240 +\n\n    100 * ( 100 * df_sub['8'] + 40 * df_sub['11'] + 30 * df_sub['7'] + 30 * df_sub['15'] + 5 * df_sub['3']+ 10 * df_sub['10']) / 215 +\n    1 * (  5 * df_sub['14'] + 1 * df_sub['4']) / 6 + 40 * df_sub['13'] ) / 241 )","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:02:54.843562Z","iopub.execute_input":"2022-08-10T07:02:54.843943Z","iopub.status.idle":"2022-08-10T07:02:54.924015Z","shell.execute_reply.started":"2022-08-10T07:02:54.843912Z","shell.execute_reply":"2022-08-10T07:02:54.923090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:26:20.483832Z","iopub.execute_input":"2022-08-10T06:26:20.484303Z","iopub.status.idle":"2022-08-10T06:26:21.614036Z","shell.execute_reply.started":"2022-08-10T06:26:20.484243Z","shell.execute_reply":"2022-08-10T06:26:21.612737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit['prediction'] = df_sub['weighted_avg'].tolist()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:02:58.128968Z","iopub.execute_input":"2022-08-10T07:02:58.129561Z","iopub.status.idle":"2022-08-10T07:02:58.287909Z","shell.execute_reply.started":"2022-08-10T07:02:58.129520Z","shell.execute_reply":"2022-08-10T07:02:58.286917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:03:02.059520Z","iopub.execute_input":"2022-08-10T07:03:02.059968Z","iopub.status.idle":"2022-08-10T07:03:05.339432Z","shell.execute_reply.started":"2022-08-10T07:03:02.059903Z","shell.execute_reply":"2022-08-10T07:03:05.338206Z"},"trusted":true},"execution_count":null,"outputs":[]}]}