{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.core.display import display, HTML, Javascript\n\n# ----- Notebook Theme -----\ncolor_map = ['#16a085', '#e8f6f3', '#d0ece7', '#a2d9ce', '#73c6b6', '#45b39d', \n                        '#16a085', '#138d75', '#117a65', '#0e6655', '#0b5345']\n\nprompt = color_map[-1]\nmain_color = color_map[0]\nstrong_main_color = color_map[1]\ncustom_colors = [strong_main_color, main_color]\n\ncss_file = ''' \n\ndiv #notebook {\nbackground-color: white;\nline-height: 20px;\n}\n\n#notebook-container {\n%s\nmargin-top: 2em;\npadding-top: 2em;\nborder-top: 4px solid %s; /* light orange */\n-webkit-box-shadow: 0px 0px 8px 2px rgba(224, 212, 226, 0.5); /* pink */\n    box-shadow: 0px 0px 8px 2px rgba(224, 212, 226, 0.5); /* pink */\n}\n\ndiv .input {\nmargin-bottom: 1em;\n}\n\n.rendered_html h1, .rendered_html h2, .rendered_html h3, .rendered_html h4, .rendered_html h5, .rendered_html h6 {\ncolor: %s; /* light orange */\nfont-weight: 600;\n}\n\ndiv.input_area {\nborder: none;\n    background-color: %s; /* rgba(229, 143, 101, 0.1); light orange [exactly #E58F65] */\n    border-top: 2px solid %s; /* light orange */\n}\n\ndiv.input_prompt {\ncolor: %s; /* light blue */\n}\n\ndiv.output_prompt {\ncolor: %s; /* strong orange */\n}\n\ndiv.cell.selected:before, div.cell.selected.jupyter-soft-selected:before {\nbackground: %s; /* light orange */\n}\n\ndiv.cell.selected, div.cell.selected.jupyter-soft-selected {\n    border-color: %s; /* light orange */\n}\n\n.edit_mode div.cell.selected:before {\nbackground: %s; /* light orange */\n}\n\n.edit_mode div.cell.selected {\nborder-color: %s; /* light orange */\n\n}\n'''\ndef to_rgb(h): \n    return tuple(int(h[i:i+2], 16) for i in [0, 2, 4])\n\nmain_color_rgba = 'rgba(%s, %s, %s, 0.1)' % (to_rgb(main_color[1:]))\nopen('notebook.css', 'w').write(css_file % ('width: 95%;', main_color, main_color, main_color_rgba, main_color,  main_color, prompt, main_color, main_color, main_color, main_color))\n\ndef nb(): \n    return HTML(\"<style>\" + open(\"notebook.css\", \"r\").read() + \"</style>\")\nnb()","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"papermill":{"duration":0.080811,"end_time":"2022-05-10T22:31:33.572169","exception":false,"start_time":"2022-05-10T22:31:33.491358","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T13:21:09.865849Z","iopub.execute_input":"2022-06-06T13:21:09.866499Z","iopub.status.idle":"2022-06-06T13:21:09.913726Z","shell.execute_reply.started":"2022-06-06T13:21:09.866370Z","shell.execute_reply":"2022-06-06T13:21:09.912581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<img src=\"https://github.com/AILab-MLTools/LightAutoML/raw/master/imgs/LightAutoML_logo_big.png\" alt=\"LightAutoML logo\" style=\"width:70%;\"/>","metadata":{"papermill":{"duration":0.055031,"end_time":"2022-05-10T22:31:33.681191","exception":false,"start_time":"2022-05-10T22:31:33.62616","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# LightAutoML baseline\n\nOfficial LightAutoML github repository is [here](https://github.com/AILab-MLTools/LightAutoML). \n\n### Do not forget to put upvote for the notebook and the ⭐️ for github repo if you like it - one click for you, great pleasure for us ☺️ ","metadata":{"papermill":{"duration":0.053669,"end_time":"2022-05-10T22:31:33.789035","exception":false,"start_time":"2022-05-10T22:31:33.735366","status":"completed"},"tags":[]}},{"cell_type":"code","source":"s = '<iframe src=\"https://ghbtns.com/github-btn.html?user=sb-ai-lab&repo=LightAutoML&type=star&count=true&size=large\" frameborder=\"0\" scrolling=\"0\" width=\"170\" height=\"30\" title=\"LightAutoML GitHub\"></iframe>'\nHTML(s)","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.06666,"end_time":"2022-05-10T22:31:33.910111","exception":false,"start_time":"2022-05-10T22:31:33.843451","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T13:21:14.612932Z","iopub.execute_input":"2022-06-06T13:21:14.613330Z","iopub.status.idle":"2022-06-06T13:21:14.623086Z","shell.execute_reply.started":"2022-06-06T13:21:14.613300Z","shell.execute_reply":"2022-06-06T13:21:14.622049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## This notebook is the updated copy of our [Tutorial_1 from the GIT repository](https://github.com/AILab-MLTools/LightAutoML/blob/master/examples/tutorials/Tutorial_1_basics.ipynb). Please check our [tutorials folder](https://github.com/AILab-MLTools/LightAutoML/blob/master/examples/tutorials) if you are interested in other examples of LightAutoML functionality.","metadata":{"papermill":{"duration":0.055266,"end_time":"2022-05-10T22:31:34.021507","exception":false,"start_time":"2022-05-10T22:31:33.966241","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## 0. Prerequisites","metadata":{"papermill":{"duration":0.055424,"end_time":"2022-05-10T22:31:34.133584","exception":false,"start_time":"2022-05-10T22:31:34.07816","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### 0.0. install LightAutoML","metadata":{"papermill":{"duration":0.053949,"end_time":"2022-05-10T22:31:34.241795","exception":false,"start_time":"2022-05-10T22:31:34.187846","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%capture\n!pip3 install -U lightautoml\n\n# QUICK WORKAROUND FOR PROBLEM WITH PANDAS\n!pip3 install -U pandas","metadata":{"_kg_hide-output":true,"papermill":{"duration":132.884228,"end_time":"2022-05-10T22:33:47.180161","exception":false,"start_time":"2022-05-10T22:31:34.295933","status":"completed"},"scrolled":true,"tags":[],"execution":{"iopub.status.busy":"2022-06-06T13:21:17.117561Z","iopub.execute_input":"2022-06-06T13:21:17.118211Z","iopub.status.idle":"2022-06-06T13:23:37.975257Z","shell.execute_reply.started":"2022-06-06T13:21:17.118148Z","shell.execute_reply":"2022-06-06T13:23:37.972990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 0.1. Import libraries\n\nHere we will import the libraries we use in this kernel:\n- Standard python libraries for timing, working with OS etc.\n- Essential python DS libraries like numpy, pandas, scikit-learn and torch (the last we will use in the next cell)\n- LightAutoML modules: `TabularAutoML` preset for AutoML model creation and Task class to setup what kind of ML problem we solve (binary/multiclass classification or regression)","metadata":{"papermill":{"duration":0.054045,"end_time":"2022-05-10T22:33:47.293341","exception":false,"start_time":"2022-05-10T22:33:47.239296","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Standard python libraries\nimport os\nimport time\n\n# Essential DS libraries\nimport numpy as np\nimport pandas as pd\nimport torch\n\n# LightAutoML presets, task and report generation\nfrom lightautoml.automl.presets.tabular_presets import TabularAutoML\nfrom lightautoml.tasks import Task","metadata":{"papermill":{"duration":8.870692,"end_time":"2022-05-10T22:33:56.218407","exception":false,"start_time":"2022-05-10T22:33:47.347715","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T13:24:08.578674Z","iopub.execute_input":"2022-06-06T13:24:08.579224Z","iopub.status.idle":"2022-06-06T13:24:11.576728Z","shell.execute_reply.started":"2022-06-06T13:24:08.579168Z","shell.execute_reply":"2022-06-06T13:24:11.575614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 0.2. Constants\n\nHere we setup the constants to use in the kernel:\n- `N_THREADS` - number of vCPUs for LightAutoML model creation\n- `N_FOLDS` - number of folds in LightAutoML inner CV\n- `RANDOM_STATE` - random seed for better reproducibility\n- `TEST_SIZE` - houldout data part size \n- `TIMEOUT` - limit in seconds for model to train\n- `TARGET_NAME` - target column name in dataset","metadata":{"papermill":{"duration":0.055772,"end_time":"2022-05-10T22:33:56.330791","exception":false,"start_time":"2022-05-10T22:33:56.275019","status":"completed"},"tags":[]}},{"cell_type":"code","source":"N_THREADS = 4\nN_FOLDS = 5\nRANDOM_STATE = 42\nTEST_SIZE = 0.2\nTIMEOUT = 8 * 3600 # equal to 8 hours\nTARGET_NAME = 'target'","metadata":{"papermill":{"duration":0.063297,"end_time":"2022-05-10T22:33:56.449182","exception":false,"start_time":"2022-05-10T22:33:56.385885","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T13:24:16.142960Z","iopub.execute_input":"2022-06-06T13:24:16.143682Z","iopub.status.idle":"2022-06-06T13:24:16.149596Z","shell.execute_reply.started":"2022-06-06T13:24:16.143638Z","shell.execute_reply":"2022-06-06T13:24:16.148367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 0.3. Imported models setup\n\nFor better reproducibility fix numpy random seed with max number of threads for Torch (which usually try to use all the threads on server):","metadata":{"papermill":{"duration":0.054426,"end_time":"2022-05-10T22:33:56.558664","exception":false,"start_time":"2022-05-10T22:33:56.504238","status":"completed"},"tags":[]}},{"cell_type":"code","source":"np.random.seed(RANDOM_STATE)\ntorch.set_num_threads(N_THREADS)","metadata":{"papermill":{"duration":0.102128,"end_time":"2022-05-10T22:33:56.715858","exception":false,"start_time":"2022-05-10T22:33:56.61373","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T13:24:16.992934Z","iopub.execute_input":"2022-06-06T13:24:16.993321Z","iopub.status.idle":"2022-06-06T13:24:17.034123Z","shell.execute_reply.started":"2022-06-06T13:24:16.993288Z","shell.execute_reply":"2022-06-06T13:24:17.033141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 0.4. Data loading\n\nFor now it's time to load the data:","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_pickle('../input/amexaggdatapicklef32/train_agg_f32.pkl', compression=\"gzip\")\nprint(train_data.shape)\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:29:15.144142Z","iopub.execute_input":"2022-06-06T15:29:15.144742Z","iopub.status.idle":"2022-06-06T15:29:36.477762Z","shell.execute_reply.started":"2022-06-06T15:29:15.144693Z","shell.execute_reply":"2022-06-06T15:29:36.476385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### In the cell below we used the trick with data denoising proposed by [@RADDAR](https://www.kaggle.com/code/raddar/the-data-has-random-uniform-noise-added/notebook) and [Chris](https://www.kaggle.com/competitions/amex-default-prediction/discussion/327651) - upvote their notebook and discussion topic for the great insight 👍","metadata":{}},{"cell_type":"code","source":"for col in train_data.columns:\n    if train_data[col].dtype=='float16':\n        train_data[col] = train_data[col].astype('float32').round(decimals=2).astype('float16')","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:29:49.292733Z","iopub.execute_input":"2022-06-06T15:29:49.293424Z","iopub.status.idle":"2022-06-06T15:29:55.519125Z","shell.execute_reply.started":"2022-06-06T15:29:49.293377Z","shell.execute_reply":"2022-06-06T15:29:55.517382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 0.5. OOF and test predictions from XGB kernel\n\n#### In cell below we upload predictions for train and test datasets from [XGB Starter kernel](https://www.kaggle.com/code/cdeotte/xgboost-starter-0-793) made by [@Chris](https://www.kaggle.com/cdeotte) - if you still didn't upvote it, that's a great chance 👍:","metadata":{}},{"cell_type":"code","source":"oof_mapper = pd.read_csv('../input/xgboost-starter-0-793/oof_xgb_v1.csv').set_index('customer_ID')\ntest_mapper = pd.read_csv('../input/xgboost-starter-0-793/submission_xgb_v1.csv').set_index('customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-06-06T13:25:41.859994Z","iopub.execute_input":"2022-06-06T13:25:41.860393Z","iopub.status.idle":"2022-06-06T13:25:45.019349Z","shell.execute_reply.started":"2022-06-06T13:25:41.860362Z","shell.execute_reply":"2022-06-06T13:25:45.018184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chris_xgb_oof = train_data['customer_ID'].map(oof_mapper['oof_pred']).values\nchris_xgb_oof","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:30:10.061899Z","iopub.execute_input":"2022-06-06T15:30:10.063027Z","iopub.status.idle":"2022-06-06T15:30:10.904198Z","shell.execute_reply.started":"2022-06-06T15:30:10.062965Z","shell.execute_reply":"2022-06-06T15:30:10.902232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Task definition","metadata":{"papermill":{"duration":0.13147,"end_time":"2022-05-11T03:46:52.221279","exception":false,"start_time":"2022-05-11T03:46:52.089809","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### 1.1. Task type\n\nOn the cell below we create Task object - the class to setup what task LightAutoML model should solve with specific loss and metric if necessary (more info can be found [here](https://lightautoml.readthedocs.io/en/latest/pages/modules/generated/lightautoml.tasks.base.Task.html#lightautoml.tasks.base.Task) in our documentation):","metadata":{"papermill":{"duration":0.13262,"end_time":"2022-05-11T03:46:52.514656","exception":false,"start_time":"2022-05-11T03:46:52.382036","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# COMPETITION METRIC FROM Konstantin Yakovlev\n# https://www.kaggle.com/kyakovlev\n# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327534\ndef amex_metric_mod(y_true, y_pred):\n\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T13:25:59.110560Z","iopub.execute_input":"2022-06-06T13:25:59.111003Z","iopub.status.idle":"2022-06-06T13:25:59.122690Z","shell.execute_reply.started":"2022-06-06T13:25:59.110965Z","shell.execute_reply":"2022-06-06T13:25:59.121589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"task = Task('binary', )","metadata":{"papermill":{"duration":0.142447,"end_time":"2022-05-11T03:46:52.788728","exception":false,"start_time":"2022-05-11T03:46:52.646281","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T13:25:59.871481Z","iopub.execute_input":"2022-06-06T13:25:59.871911Z","iopub.status.idle":"2022-06-06T13:25:59.883370Z","shell.execute_reply.started":"2022-06-06T13:25:59.871874Z","shell.execute_reply":"2022-06-06T13:25:59.882259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2. Feature roles setup","metadata":{"papermill":{"duration":0.135627,"end_time":"2022-05-11T03:46:53.056301","exception":false,"start_time":"2022-05-11T03:46:52.920674","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"To solve the task, we need to setup columns roles. The **only role you must setup is target role**, everything else (drop, numeric, categorical, group, weights etc.) is up to user - LightAutoML models have automatic columns typization inside:","metadata":{"papermill":{"duration":0.131703,"end_time":"2022-05-11T03:46:53.320032","exception":false,"start_time":"2022-05-11T03:46:53.188329","status":"completed"},"tags":[]}},{"cell_type":"code","source":"roles = {\n    'target': TARGET_NAME,\n    'drop': ['customer_ID']\n}","metadata":{"papermill":{"duration":0.139165,"end_time":"2022-05-11T03:46:53.5898","exception":false,"start_time":"2022-05-11T03:46:53.450635","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T13:26:01.741659Z","iopub.execute_input":"2022-06-06T13:26:01.742394Z","iopub.status.idle":"2022-06-06T13:26:01.746771Z","shell.execute_reply.started":"2022-06-06T13:26:01.742357Z","shell.execute_reply":"2022-06-06T13:26:01.746109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.3. LightAutoML model creation - TabularAutoML preset","metadata":{"papermill":{"duration":0.130994,"end_time":"2022-05-11T03:46:53.853469","exception":false,"start_time":"2022-05-11T03:46:53.722475","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"In next the cell we are going to create LightAutoML model with `TabularAutoML` class - preset with default model structure like in the image below:\n\n<img src=\"https://github.com/AILab-MLTools/LightAutoML/raw/master/imgs/tutorial_blackbox_pipeline.png\" alt=\"TabularAutoML preset pipeline\" style=\"width:85%;\"/>\n\nin just several lines. Let's discuss the params we can setup:\n- `task` - the type of the ML task (the only **must have** parameter)\n- `timeout` - time limit in seconds for model to train\n- `cpu_limit` - vCPU count for model to use\n- `reader_params` - parameter change for Reader object inside preset, which works on the first step of data preparation: automatic feature typization, preliminary almost-constant features, correct CV setup etc. For example, we setup `n_jobs` threads for typization algo, `cv` folds and `random_state` as inside CV seed.\n\n**Important note**: `reader_params` key is one of the YAML config keys, which is used inside `TabularAutoML` preset. [More details](https://github.com/AILab-MLTools/LightAutoML/blob/master/lightautoml/automl/presets/tabular_config.yml) on its structure with explanation comments can be found on the link attached. Each key from this config can be modified with user settings during preset object initialization. To get more info about different parameters setting (for example, ML algos which can be used in `general_params->use_algos`) please take a look at our [article on TowardsDataScience](https://towardsdatascience.com/lightautoml-preset-usage-tutorial-2cce7da6f936).\n\nMoreover, to receive the automatic report for our model we can use `ReportDeco` decorator and work with the decorated version in the same way as we do with usual one (more details in [this tutorial](https://github.com/AILab-MLTools/LightAutoML/blob/master/examples/tutorials/Tutorial_1_basics.ipynb))","metadata":{"papermill":{"duration":0.132328,"end_time":"2022-05-11T03:46:54.115757","exception":false,"start_time":"2022-05-11T03:46:53.983429","status":"completed"},"tags":[]}},{"cell_type":"code","source":"automl = TabularAutoML(\n    task = task, \n    timeout = TIMEOUT,\n    cpu_limit = N_THREADS,\n    general_params = {'use_algos': [['linear_l2', 'lgb', 'cb']]},\n    reader_params = {'n_jobs': 1, 'cv': N_FOLDS, 'random_state': RANDOM_STATE},\n    selection_params = {'mode': 0}\n)","metadata":{"papermill":{"duration":0.179009,"end_time":"2022-05-11T03:46:54.42478","exception":false,"start_time":"2022-05-11T03:46:54.245771","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T13:26:05.080864Z","iopub.execute_input":"2022-06-06T13:26:05.081512Z","iopub.status.idle":"2022-06-06T13:26:05.107357Z","shell.execute_reply.started":"2022-06-06T13:26:05.081470Z","shell.execute_reply":"2022-06-06T13:26:05.106575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. AutoML training","metadata":{"papermill":{"duration":0.133697,"end_time":"2022-05-11T03:46:54.690292","exception":false,"start_time":"2022-05-11T03:46:54.556595","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"To run autoML training use fit_predict method:\n- `train_data` - Dataset to train.\n- `roles` - Roles dict.\n- `verbose` - Controls the verbosity: the higher, the more messages.\n        <1  : messages are not displayed;\n        >=1 : the computation process for layers is displayed;\n        >=2 : the information about folds processing is also displayed;\n        >=3 : the hyperparameters optimization process is also displayed;\n        >=4 : the training process for every algorithm is displayed;\n\nNote: out-of-fold prediction is calculated during training and returned from the fit_predict method","metadata":{"papermill":{"duration":0.131342,"end_time":"2022-05-11T03:46:54.952589","exception":false,"start_time":"2022-05-11T03:46:54.821247","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time \noof_pred = automl.fit_predict(train_data, roles = roles, verbose = 3)","metadata":{"_kg_hide-output":true,"papermill":{"duration":1100.343738,"end_time":"2022-05-11T04:05:15.432529","exception":false,"start_time":"2022-05-11T03:46:55.088791","status":"completed"},"scrolled":true,"tags":[],"execution":{"iopub.status.busy":"2022-06-06T13:26:07.017219Z","iopub.execute_input":"2022-06-06T13:26:07.017890Z","iopub.status.idle":"2022-06-06T14:37:57.580868Z","shell.execute_reply.started":"2022-06-06T13:26:07.017850Z","shell.execute_reply":"2022-06-06T14:37:57.579764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(automl.create_model_str_desc())","metadata":{"papermill":{"duration":0.183914,"end_time":"2022-05-11T04:05:15.786503","exception":false,"start_time":"2022-05-11T04:05:15.602589","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T14:37:57.583552Z","iopub.execute_input":"2022-06-06T14:37:57.583950Z","iopub.status.idle":"2022-06-06T14:37:57.590697Z","shell.execute_reply.started":"2022-06-06T14:37:57.583913Z","shell.execute_reply":"2022-06-06T14:37:57.589449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'OOF score: {amex_metric_mod(train_data[TARGET_NAME].values, oof_pred.data[:, 0])}')","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:37:57.592208Z","iopub.execute_input":"2022-06-06T14:37:57.592534Z","iopub.status.idle":"2022-06-06T14:37:57.809920Z","shell.execute_reply.started":"2022-06-06T14:37:57.592504Z","shell.execute_reply":"2022-06-06T14:37:57.808500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_w = None\nbest_sc = -1\nfor w in np.arange(0, 1.01, 0.01):\n    sc = amex_metric_mod(train_data[TARGET_NAME].values, w * oof_pred.data[:, 0] + (1-w)*chris_xgb_oof)\n    if sc > best_sc:\n        best_sc = sc\n        best_w = w\n        print('{:.7f} {:.2f}'.format(best_sc, best_w))\n        \nprint('Finally selected: Score = {:.7f}, Best_w = {:.2f}'.format(best_sc, best_w))","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:39:57.900649Z","iopub.execute_input":"2022-06-06T15:39:57.901387Z","iopub.status.idle":"2022-06-06T15:40:22.015424Z","shell.execute_reply.started":"2022-06-06T15:39:57.901346Z","shell.execute_reply":"2022-06-06T15:40:22.014179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Final OOF score: {amex_metric_mod(train_data[TARGET_NAME].values, best_w * oof_pred.data[:, 0] + (1-best_w)*chris_xgb_oof)}')","metadata":{"execution":{"iopub.status.busy":"2022-06-06T15:34:06.518524Z","iopub.execute_input":"2022-06-06T15:34:06.519212Z","iopub.status.idle":"2022-06-06T15:34:06.750010Z","shell.execute_reply.started":"2022-06-06T15:34:06.519162Z","shell.execute_reply":"2022-06-06T15:34:06.748933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Feature importances calculation \n\nFor feature importances calculation we have 2 different methods in LightAutoML:\n- Fast (`fast`) - this method uses feature importances from feature selector LGBM model inside LightAutoML. It works extremely fast and almost always (almost because of situations, when feature selection is turned off or selector was removed from the final models with all GBM models). no need to use new labelled data.\n- Accurate (`accurate`) - this method calculate *features permutation importances* for the whole LightAutoML model based on the **new labelled data**. It always works but can take a lot of time to finish (depending on the model structure, new labelled dataset size etc.).","metadata":{"papermill":{"duration":0.168756,"end_time":"2022-05-11T04:05:16.862165","exception":false,"start_time":"2022-05-11T04:05:16.693409","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# %%time\n\n# # Fast feature importances calculation\n# fast_fi = automl.get_feature_scores('fast').head(75)\n# top_3_features = fast_fi['Feature'].values[:3]\n# fast_fi.set_index('Feature')['Importance'].plot.bar(figsize = (30, 10), grid = True)","metadata":{"papermill":{"duration":1.474733,"end_time":"2022-05-11T04:05:18.509011","exception":false,"start_time":"2022-05-11T04:05:17.034278","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T14:37:57.812758Z","iopub.execute_input":"2022-06-06T14:37:57.813230Z","iopub.status.idle":"2022-06-06T14:37:57.820803Z","shell.execute_reply.started":"2022-06-06T14:37:57.813182Z","shell.execute_reply":"2022-06-06T14:37:57.818861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fast_fi.head()","metadata":{"papermill":{"duration":0.185287,"end_time":"2022-05-11T04:05:18.864914","exception":false,"start_time":"2022-05-11T04:05:18.679627","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T14:37:57.823048Z","iopub.execute_input":"2022-06-06T14:37:57.823509Z","iopub.status.idle":"2022-06-06T14:37:57.837762Z","shell.execute_reply.started":"2022-06-06T14:37:57.823466Z","shell.execute_reply":"2022-06-06T14:37:57.836610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Plot PDP graphs for LightAutoML model","metadata":{}},{"cell_type":"code","source":"data = pd.read_pickle('../input/amexaggdatapicklef32/test_agg_f32_part_0.pkl', compression=\"gzip\")\nfor col in data.columns:\n    if data[col].dtype=='float16':\n        data[col] = data[col].astype('float32').round(decimals=2).astype('float16')","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:37:57.839114Z","iopub.execute_input":"2022-06-06T14:37:57.839562Z","iopub.status.idle":"2022-06-06T14:38:05.622878Z","shell.execute_reply.started":"2022-06-06T14:37:57.839531Z","shell.execute_reply":"2022-06-06T14:38:05.621851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"automl.plot_pdp(data.sample(20000), feature_name='P_2_last')","metadata":{"papermill":{"duration":49.864486,"end_time":"2022-05-11T04:06:08.900465","exception":false,"start_time":"2022-05-11T04:05:19.035979","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T14:42:39.910962Z","iopub.execute_input":"2022-06-06T14:42:39.911298Z","iopub.status.idle":"2022-06-06T14:47:11.727134Z","shell.execute_reply.started":"2022-06-06T14:42:39.911267Z","shell.execute_reply":"2022-06-06T14:47:11.726035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"automl.plot_pdp(data.sample(20000), feature_name='D_39_last')","metadata":{"papermill":{"duration":34.892412,"end_time":"2022-05-11T04:06:43.97871","exception":false,"start_time":"2022-05-11T04:06:09.086298","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T14:47:11.728701Z","iopub.execute_input":"2022-06-06T14:47:11.729423Z","iopub.status.idle":"2022-06-06T14:48:54.567633Z","shell.execute_reply.started":"2022-06-06T14:47:11.729382Z","shell.execute_reply":"2022-06-06T14:48:54.566576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Predict for test dataset\n\nWe are also ready to predict for our test competition dataset and submission file creation:","metadata":{"papermill":{"duration":0.223583,"end_time":"2022-05-11T04:07:19.850624","exception":false,"start_time":"2022-05-11T04:07:19.627041","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import gc\ndel train_data\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:48:54.570949Z","iopub.execute_input":"2022-06-06T14:48:54.571459Z","iopub.status.idle":"2022-06-06T14:48:54.746229Z","shell.execute_reply.started":"2022-06-06T14:48:54.571408Z","shell.execute_reply":"2022-06-06T14:48:54.745398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = []\nfor i in range(10):\n    data = pd.read_pickle('../input/amexaggdatapicklef32/test_agg_f32_part_{}.pkl'.format(i), compression=\"gzip\")\n    chris_xgb_test = data['customer_ID'].map(test_mapper['prediction']).values\n    for col in data.columns:\n        if data[col].dtype=='float16':\n            data[col] = data[col].astype('float32').round(decimals=2).astype('float16')\n    print(i, data.shape)\n    test_pred = automl.predict(data)\n    test_predictions += list(best_w * test_pred.data[:, 0] + (1-best_w)*chris_xgb_test)","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:48:54.747515Z","iopub.execute_input":"2022-06-06T14:48:54.748266Z","iopub.status.idle":"2022-06-06T14:54:32.987928Z","shell.execute_reply.started":"2022-06-06T14:48:54.748229Z","shell.execute_reply":"2022-06-06T14:54:32.986503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nprint(submission.shape)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-06T14:54:32.989895Z","iopub.execute_input":"2022-06-06T14:54:32.990394Z","iopub.status.idle":"2022-06-06T14:54:34.960677Z","shell.execute_reply.started":"2022-06-06T14:54:32.990343Z","shell.execute_reply":"2022-06-06T14:54:34.959921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['prediction'] = test_predictions\nsubmission.to_csv('lightautoml_tabularautoml.csv', index = False)\nsubmission","metadata":{"papermill":{"duration":2.224661,"end_time":"2022-05-11T04:08:29.643539","exception":false,"start_time":"2022-05-11T04:08:27.418878","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-06T14:54:34.961972Z","iopub.execute_input":"2022-06-06T14:54:34.962514Z","iopub.status.idle":"2022-06-06T14:54:40.251942Z","shell.execute_reply.started":"2022-06-06T14:54:34.962480Z","shell.execute_reply":"2022-06-06T14:54:40.250904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Additional materials","metadata":{"papermill":{"duration":0.218069,"end_time":"2022-05-11T04:08:30.076645","exception":false,"start_time":"2022-05-11T04:08:29.858576","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"- [Official LightAutoML github repo](https://github.com/AILab-MLTools/LightAutoML)\n- [LightAutoML documentation](https://lightautoml.readthedocs.io/en/latest)\n- [LightAutoML tutorials](https://github.com/AILab-MLTools/LightAutoML/tree/master/examples/tutorials)\n- LightAutoML course:\n    - [Part 1 - general overview](https://ods.ai/tracks/automl-course-part1) \n    - [Part 2 - LightAutoML specific applications](https://ods.ai/tracks/automl-course-part2)\n    - [Part 3 - LightAutoML customization](https://ods.ai/tracks/automl-course-part3)\n- [OpenDataScience AutoML benchmark leaderboard](https://ods.ai/competitions/automl-benchmark/leaderboard)","metadata":{"papermill":{"duration":0.223516,"end_time":"2022-05-11T04:08:30.534611","exception":false,"start_time":"2022-05-11T04:08:30.311095","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### If you still like the notebook, do not forget to put upvote for the notebook and the ⭐️ for github repo if you like it using the button below - one click for you, great pleasure for us ☺️","metadata":{"papermill":{"duration":0.217296,"end_time":"2022-05-11T04:08:30.965644","exception":false,"start_time":"2022-05-11T04:08:30.748348","status":"completed"},"tags":[]}},{"cell_type":"code","source":"s = '<iframe src=\"https://ghbtns.com/github-btn.html?user=sb-ai-lab&repo=LightAutoML&type=star&count=true&size=large\" frameborder=\"0\" scrolling=\"0\" width=\"170\" height=\"30\" title=\"LightAutoML GitHub\"></iframe>'\nHTML(s)","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2022-05-11T04:08:31.397984Z","iopub.status.busy":"2022-05-11T04:08:31.397664Z","iopub.status.idle":"2022-05-11T04:08:31.404857Z","shell.execute_reply":"2022-05-11T04:08:31.403877Z"},"papermill":{"duration":0.227298,"end_time":"2022-05-11T04:08:31.407672","exception":false,"start_time":"2022-05-11T04:08:31.180374","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}