{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.core.display import display, HTML, Javascript\n\n# ----- Notebook Theme -----\ncolor_map = ['#16a085', '#e8f6f3', '#d0ece7', '#a2d9ce', '#73c6b6', '#45b39d', \n                        '#16a085', '#138d75', '#117a65', '#0e6655', '#0b5345']\n\nprompt = color_map[-1]\nmain_color = color_map[0]\nstrong_main_color = color_map[1]\ncustom_colors = [strong_main_color, main_color]\n\ncss_file = ''' \n\ndiv #notebook {\nbackground-color: white;\nline-height: 20px;\n}\n\n#notebook-container {\n%s\nmargin-top: 2em;\npadding-top: 2em;\nborder-top: 4px solid %s; /* light orange */\n-webkit-box-shadow: 0px 0px 8px 2px rgba(224, 212, 226, 0.5); /* pink */\n    box-shadow: 0px 0px 8px 2px rgba(224, 212, 226, 0.5); /* pink */\n}\n\ndiv .input {\nmargin-bottom: 1em;\n}\n\n.rendered_html h1, .rendered_html h2, .rendered_html h3, .rendered_html h4, .rendered_html h5, .rendered_html h6 {\ncolor: %s; /* light orange */\nfont-weight: 600;\n}\n\ndiv.input_area {\nborder: none;\n    background-color: %s; /* rgba(229, 143, 101, 0.1); light orange [exactly #E58F65] */\n    border-top: 2px solid %s; /* light orange */\n}\n\ndiv.input_prompt {\ncolor: %s; /* light blue */\n}\n\ndiv.output_prompt {\ncolor: %s; /* strong orange */\n}\n\ndiv.cell.selected:before, div.cell.selected.jupyter-soft-selected:before {\nbackground: %s; /* light orange */\n}\n\ndiv.cell.selected, div.cell.selected.jupyter-soft-selected {\n    border-color: %s; /* light orange */\n}\n\n.edit_mode div.cell.selected:before {\nbackground: %s; /* light orange */\n}\n\n.edit_mode div.cell.selected {\nborder-color: %s; /* light orange */\n\n}\n'''\ndef to_rgb(h): \n    return tuple(int(h[i:i+2], 16) for i in [0, 2, 4])\n\nmain_color_rgba = 'rgba(%s, %s, %s, 0.1)' % (to_rgb(main_color[1:]))\nopen('notebook.css', 'w').write(css_file % ('width: 95%;', main_color, main_color, main_color_rgba, main_color,  main_color, prompt, main_color, main_color, main_color, main_color))\n\ndef nb(): \n    return HTML(\"<style>\" + open(\"notebook.css\", \"r\").read() + \"</style>\")\nnb()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.033589,"end_time":"2022-09-22T10:39:13.499294","exception":false,"start_time":"2022-09-22T10:39:13.465705","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-03T20:25:09.769058Z","iopub.execute_input":"2022-10-03T20:25:09.770086Z","iopub.status.idle":"2022-10-03T20:25:09.817721Z","shell.execute_reply.started":"2022-10-03T20:25:09.769953Z","shell.execute_reply":"2022-10-03T20:25:09.816310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<img src=\"https://raw.githubusercontent.com/AILab-MLTools/LightAutoML/master/imgs/LightAutoML_logo_big.png\" alt=\"LightAutoML logo\" style=\"width:70%;\"/>","metadata":{"papermill":{"duration":0.008116,"end_time":"2022-09-22T10:39:13.516051","exception":false,"start_time":"2022-09-22T10:39:13.507935","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# LightAutoML baseline\n\nOfficial LightAutoML github repository is [here](https://github.com/AILab-MLTools/LightAutoML). \n\n### Do not forget to put upvote for the notebook and the ⭐️ for github repo if you like it using the button below - one click for you, great pleasure for us ☺️ ","metadata":{"papermill":{"duration":0.0083,"end_time":"2022-09-22T10:39:13.533288","exception":false,"start_time":"2022-09-22T10:39:13.524988","status":"completed"},"tags":[]}},{"cell_type":"code","source":"s = '<iframe src=\"https://ghbtns.com/github-btn.html?user=AILab-MLTools&repo=LightAutoML&type=star&count=true&size=large\" frameborder=\"0\" scrolling=\"0\" width=\"170\" height=\"30\" title=\"LightAutoML GitHub\"></iframe>'\nHTML(s)","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.022551,"end_time":"2022-09-22T10:39:13.564448","exception":false,"start_time":"2022-09-22T10:39:13.541897","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-03T20:25:09.821590Z","iopub.execute_input":"2022-10-03T20:25:09.822543Z","iopub.status.idle":"2022-10-03T20:25:09.832740Z","shell.execute_reply.started":"2022-10-03T20:25:09.822491Z","shell.execute_reply":"2022-10-03T20:25:09.831742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 0. Prerequisites","metadata":{"papermill":{"duration":0.007798,"end_time":"2022-09-22T10:39:13.580613","exception":false,"start_time":"2022-09-22T10:39:13.572815","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### 0.0. install LightAutoML","metadata":{"papermill":{"duration":0.00876,"end_time":"2022-09-22T10:39:13.598562","exception":false,"start_time":"2022-09-22T10:39:13.589802","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%capture\n!pip install -U lightautoml","metadata":{"_kg_hide-input":false,"_kg_hide-output":false,"papermill":{"duration":85.95746,"end_time":"2022-09-22T10:40:39.564548","exception":false,"start_time":"2022-09-22T10:39:13.607088","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-03T20:25:09.834336Z","iopub.execute_input":"2022-10-03T20:25:09.835032Z","iopub.status.idle":"2022-10-03T20:26:36.195860Z","shell.execute_reply.started":"2022-10-03T20:25:09.834983Z","shell.execute_reply":"2022-10-03T20:26:36.194234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 0.1. Import libraries\n\nHere we will import the libraries we use in this kernel:\n- Standard python libraries for timing, working with OS etc.\n- Essential python DS libraries like numpy, pandas, scikit-learn and torch (the last we will use in the next cell)\n- LightAutoML modules: presets for AutoML, task and report generation module","metadata":{"papermill":{"duration":0.008257,"end_time":"2022-09-22T10:40:39.581962","exception":false,"start_time":"2022-09-22T10:40:39.573705","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Essential DS libraries\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nimport torch\nfrom sklearn.metrics import f1_score\nimport gc\n\n\n# LightAutoML presets, task and report generation\nfrom lightautoml.automl.presets.tabular_presets import TabularAutoML, TabularUtilizedAutoML\nfrom lightautoml.tasks import Task\nfrom lightautoml.report.report_deco import ReportDeco\n\npd.set_option('display.max_columns', None)","metadata":{"papermill":{"duration":3.782753,"end_time":"2022-09-22T10:40:43.373333","exception":false,"start_time":"2022-09-22T10:40:39.590580","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-03T20:26:36.198711Z","iopub.execute_input":"2022-10-03T20:26:36.199263Z","iopub.status.idle":"2022-10-03T20:26:39.483893Z","shell.execute_reply.started":"2022-10-03T20:26:36.199220Z","shell.execute_reply":"2022-10-03T20:26:39.482755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 0.2. Constants\n\nHere we setup the constants to use in the kernel:\n- `N_THREADS` - number of vCPUs for LightAutoML model creation\n- `RANDOM_STATE` - random seed for better reproducibility\n- `TEST_SIZE` - houldout data part size \n- `TIMEOUT` - limit in seconds for model to train\n- `TARGET_NAME` - target column name in dataset\n- `N_FOLDS` - number folds for training","metadata":{"papermill":{"duration":0.00883,"end_time":"2022-09-22T10:40:43.391490","exception":false,"start_time":"2022-09-22T10:40:43.382660","status":"completed"},"tags":[]}},{"cell_type":"code","source":"N_THREADS = 4\nRANDOM_STATE = 21\n# TEST_SIZE = 0.2\nTIMEOUT = 4 * 3600\nTARGET_NAME_A = 'team_A_scoring_within_10sec'\nTARGET_NAME_B = 'team_B_scoring_within_10sec'\n# N_FOLDS = 15","metadata":{"papermill":{"duration":0.017714,"end_time":"2022-09-22T10:40:43.417853","exception":false,"start_time":"2022-09-22T10:40:43.400139","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-03T20:26:39.485289Z","iopub.execute_input":"2022-10-03T20:26:39.485766Z","iopub.status.idle":"2022-10-03T20:26:39.492904Z","shell.execute_reply.started":"2022-10-03T20:26:39.485720Z","shell.execute_reply":"2022-10-03T20:26:39.490750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 0.3. Imported models setup\n\nFor better reproducibility fix numpy random seed with max number of threads for Torch (which usually try to use all the threads on server):","metadata":{"papermill":{"duration":0.008678,"end_time":"2022-09-22T10:40:43.436111","exception":false,"start_time":"2022-09-22T10:40:43.427433","status":"completed"},"tags":[]}},{"cell_type":"code","source":"np.random.seed(RANDOM_STATE)\ntorch.set_num_threads(N_THREADS)","metadata":{"papermill":{"duration":0.046845,"end_time":"2022-09-22T10:40:43.492083","exception":false,"start_time":"2022-09-22T10:40:43.445238","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-03T20:26:39.495062Z","iopub.execute_input":"2022-10-03T20:26:39.496109Z","iopub.status.idle":"2022-10-03T20:26:39.561944Z","shell.execute_reply.started":"2022-10-03T20:26:39.496053Z","shell.execute_reply":"2022-10-03T20:26:39.560644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 0.4. Data loading\nLet's check the data we have:","metadata":{"papermill":{"duration":0.009278,"end_time":"2022-09-22T10:40:43.512154","exception":false,"start_time":"2022-09-22T10:40:43.502876","status":"completed"},"tags":[]}},{"cell_type":"code","source":"INPUT_DIR = Path('/kaggle/input/tabular-playground-series-oct-2022/')","metadata":{"papermill":{"duration":0.019119,"end_time":"2022-09-22T10:40:43.540221","exception":false,"start_time":"2022-09-22T10:40:43.521102","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-10-03T20:26:39.563388Z","iopub.execute_input":"2022-10-03T20:26:39.564169Z","iopub.status.idle":"2022-10-03T20:26:39.579173Z","shell.execute_reply.started":"2022-10-03T20:26:39.564123Z","shell.execute_reply":"2022-10-03T20:26:39.577907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dtypes_df = pd.read_csv(f'{INPUT_DIR}/train_dtypes.csv')\ntrain_dtypes = {k: v for (k, v) in zip(train_dtypes_df.column, train_dtypes_df.dtype)}\ntest_dtypes_df = pd.read_csv(f'{INPUT_DIR}/test_dtypes.csv')\ntest_dtypes = {k: v for (k, v) in zip(test_dtypes_df.column, test_dtypes_df.dtype)}","metadata":{"execution":{"iopub.status.busy":"2022-10-03T20:26:39.580460Z","iopub.execute_input":"2022-10-03T20:26:39.581001Z","iopub.status.idle":"2022-10-03T20:26:39.627503Z","shell.execute_reply.started":"2022-10-03T20:26:39.580955Z","shell.execute_reply":"2022-10-03T20:26:39.626612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.DataFrame({}, columns=train_dtypes.keys())\nfor i in range(10):\n    temp_data = pd.read_csv(f'{INPUT_DIR}/train_{i}.csv', dtype=train_dtypes)\n    temp_data = temp_data.sample(frac=0.3, random_state=RANDOM_STATE)\n    train_data = pd.concat([train_data, temp_data])\n    del temp_data\n    gc.collect()\ntrain_data = train_data.astype(train_dtypes)\ntrain_data = train_data.dropna(axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-10-03T20:26:39.632633Z","iopub.execute_input":"2022-10-03T20:26:39.633020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data.shape)\ntrain_data.head()","metadata":{"papermill":{"duration":0.152494,"end_time":"2022-09-22T10:40:43.701696","exception":false,"start_time":"2022-09-22T10:40:43.549202","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.050605,"end_time":"2022-09-22T10:40:43.761356","exception":false,"start_time":"2022-09-22T10:40:43.710751","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(INPUT_DIR / 'test.csv', dtype=test_dtypes)\ntest_data = test_data.astype(test_dtypes)\ntest_data = test_data.drop(['id'], axis=1)\nprint(test_data.shape)\ntest_data.head()","metadata":{"papermill":{"duration":0.052571,"end_time":"2022-09-22T10:40:43.822801","exception":false,"start_time":"2022-09-22T10:40:43.770230","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.info()","metadata":{"papermill":{"duration":0.028894,"end_time":"2022-09-22T10:40:43.861089","exception":false,"start_time":"2022-09-22T10:40:43.832195","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(INPUT_DIR / 'sample_submission.csv')\nprint(submission.shape)\nsubmission.head()","metadata":{"papermill":{"duration":0.045986,"end_time":"2022-09-22T10:40:43.917168","exception":false,"start_time":"2022-09-22T10:40:43.871182","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 0.5. Feature engineering\nLet's make same new features:","metadata":{}},{"cell_type":"markdown","source":"Thanks to [MARCOCIAV's](https://www.kaggle.com/code/chazzer/rocket-league-xgboost-feat-engineering-cv), [INFRARED's](https://www.kaggle.com/code/infrarosso/tps-oct-2022-eda-lgbm-model-submit), [C4RL05/V's](https://www.kaggle.com/code/cv13j0/tps-oct22-gbdt-classifier#💡-6.-Feature-Engineering) notebooks for new features. Upvote them first.","metadata":{}},{"cell_type":"code","source":"def feature_engineering(data):\n    data['distance_for_A'] = ((data['ball_pos_x']-0)**2 + (data['ball_pos_y']-100)**2 + (data['ball_pos_z']-20)**2)**0.5\n    data['distance_for_B'] = ((data['ball_pos_x']-0)**2 + (data['ball_pos_y']+100)**2 + (data['ball_pos_z']-20)**2)**0.5  \n    data['team_a_boost'] = data['p0_boost'] + data['p1_boost']+ data['p2_boost']\n    data['team_b_boost'] = data['p3_boost'] + data['p4_boost']+ data['p5_boost']\n    data['team_a_advantage'] = data['team_a_boost'] - data['team_b_boost']\n    data['boost_timer_a'] = data['boost0_timer'] + data['boost1_timer']+ data['boost2_timer']\n    data['boost_timer_b'] = data['boost3_timer'] + data['boost4_timer']+ data['boost5_timer']\n    for i in range(6):\n        data[f'p{i}_ball_distance'] = ((data['ball_pos_x']-data[f'p{i}_pos_x'])**2 + (data['ball_pos_y']-data[f'p{i}_pos_y'])**2 + (data['ball_pos_z']-data[f'p{i}_pos_z'])**2)**0.5\n        data[f'p{i}_boost_timer'] = data[f'p{i}_boost']*data[f'boost{i}_timer']\n        data[f'p{i}_pos'] = np.sqrt(data[f'p{i}_pos_x']**2 + data[f'p{i}_pos_y']**2 + data[f'p{i}_pos_z']**2)\n        data[f'p{i}_vel'] = np.sqrt(data[f'p{i}_vel_x']**2 + data[f'p{i}_vel_y']**2 + data[f'p{i}_vel_z']**2)\n    return data\n\n\ndef reduce_memory(data, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = data.memory_usage().sum() / 1024 ** 2\n    for column in data.columns:\n        column_type = data[column].dtypes\n        if column_type in numerics:\n            c_min = data[column].min()\n            c_max = data[column].max()\n            if str(column_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    data[column] = data[column].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    data[column] = data[column].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    data[column] = data[column].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    data[column] = data[column].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    data[column] = data[column].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    data[column] = data[column].astype(np.float32)\n                else:\n                    data[column] = data[column].astype(np.float64)\n    end_mem = data.memory_usage().sum() / 1024 ** 2\n    if verbose:\n        print('Memory usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100*(start_mem - end_mem) / start_mem))\n    return data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time \n\nfor data in [train_data, test_data]:\n    data = feature_engineering(data)\n    data = reduce_memory(data)\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Task definition","metadata":{"papermill":{"duration":0.00906,"end_time":"2022-09-22T10:40:43.935617","exception":false,"start_time":"2022-09-22T10:40:43.926557","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### 1.1. Task type\n\nOn the cell below we create Task object - the class to setup what task LightAutoML model should solve with specific loss and metric if necessary (more info can be found [here](https://lightautoml.readthedocs.io/en/latest/generated/lightautoml.tasks.base.Task.html#lightautoml.tasks.base.Task) in our documentation):","metadata":{"execution":{"iopub.execute_input":"2022-03-03T07:13:10.76383Z","iopub.status.busy":"2022-03-03T07:13:10.763538Z","iopub.status.idle":"2022-03-03T07:13:10.770524Z","shell.execute_reply":"2022-03-03T07:13:10.769341Z","shell.execute_reply.started":"2022-03-03T07:13:10.763803Z"},"papermill":{"duration":0.009005,"end_time":"2022-09-22T10:40:43.954105","exception":false,"start_time":"2022-09-22T10:40:43.945100","status":"completed"},"tags":[]}},{"cell_type":"code","source":"task = Task(name = 'binary',\n#             metric = lambda y_true, y_pred: f1_score(y_true, (y_pred > 0.5)*1)\n           )","metadata":{"papermill":{"duration":0.01948,"end_time":"2022-09-22T10:40:44.010690","exception":false,"start_time":"2022-09-22T10:40:43.991210","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.2. Feature roles setup\nTo solve the task, we need to setup columns roles. The **only role you must setup is target role**, everything else (drop, numeric, categorical, group, weights etc.) is up to user - LightAutoML models have automatic columns typization inside:","metadata":{"papermill":{"duration":0.008725,"end_time":"2022-09-22T10:40:44.028683","exception":false,"start_time":"2022-09-22T10:40:44.019958","status":"completed"},"tags":[]}},{"cell_type":"code","source":"roles_A = {'target': TARGET_NAME_A,\n           'drop': ['game_num',\n                    'event_id',\n                    'event_time',\n                    'player_scoring_next',\n                    'team_scoring_next',\n                    TARGET_NAME_B]\n          }\nroles_B = {'target': TARGET_NAME_B,\n           'drop': ['game_num',\n                    'event_id','event_time',\n                    'player_scoring_next',\n                    'team_scoring_next', \n                    TARGET_NAME_A]\n          }","metadata":{"papermill":{"duration":0.016958,"end_time":"2022-09-22T10:40:44.054404","exception":false,"start_time":"2022-09-22T10:40:44.037446","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1.3. LightAutoML model creation - TabularAutoML preset","metadata":{"papermill":{"duration":0.008854,"end_time":"2022-09-22T10:40:44.072415","exception":false,"start_time":"2022-09-22T10:40:44.063561","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"In next the cell we are going to create LightAutoML model with `TabularAutoML` class - preset with default model structure like in the image below:\n\n<img src=\"https://github.com/AILab-MLTools/LightAutoML/raw/master/imgs/tutorial_blackbox_pipeline.png\" alt=\"TabularAutoML preset pipeline\" style=\"width:75%;\"/>\n\nin just several lines. Let's discuss the params we can setup:\n- `task` - the type of the ML task (the only **must have** parameter)\n- `timeout` - time limit in seconds for model to train\n- `cpu_limit` - vCPU count for model to use\n- `reader_params` - parameter change for Reader object inside preset, which works on the first step of data preparation: automatic feature typization, preliminary almost-constant features, correct CV setup etc. For example, we setup `n_jobs` threads for typization algo, `cv` folds and `random_state` as inside CV seed.\n\n**Important note**: `reader_params` key is one of the YAML config keys, which is used inside `TabularAutoML` preset. [More details](https://github.com/AILab-MLTools/blob/master/lightautoml/automl/presets/tabular_config.yml) on its structure with explanation comments can be found on the link attached. Each key from this config can be modified with user settings during preset object initialization. To get more info about different parameters setting (for example, ML algos which can be used in `general_params->use_algos`) please take a look at our [article on TowardsDataScience](https://towardsdatascience.com/lightautoml-preset-usage-tutorial-2cce7da6f936).\n\nMoreover, to receive the automatic report for our model we will use `ReportDeco` decorator and work with the decorated version in the same way as we do with usual one. ","metadata":{"papermill":{"duration":0.009086,"end_time":"2022-09-22T10:40:44.090462","exception":false,"start_time":"2022-09-22T10:40:44.081376","status":"completed"},"tags":[]}},{"cell_type":"code","source":"automl_A = TabularUtilizedAutoML(task = task,\n                       timeout = TIMEOUT,\n                       cpu_limit = N_THREADS,\n                       reader_params = {'n_jobs': N_THREADS, 'random_state': RANDOM_STATE},\n#                        general_params = {'use_algos': [['linear_l2', 'lgb']]}\n                      )\n\nautoml_B = TabularUtilizedAutoML(task = task,\n                       timeout = TIMEOUT,\n                       cpu_limit = N_THREADS,\n                       reader_params = {'n_jobs': N_THREADS, 'random_state': RANDOM_STATE},\n#                        general_params = {'use_algos': [['linear_l2', 'lgb']]}\n                      )","metadata":{"papermill":{"duration":0.033711,"end_time":"2022-09-22T10:40:44.132974","exception":false,"start_time":"2022-09-22T10:40:44.099263","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. AutoML training","metadata":{"papermill":{"duration":0.009213,"end_time":"2022-09-22T10:40:44.151877","exception":false,"start_time":"2022-09-22T10:40:44.142664","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"To run autoML training use fit_predict method:\n- `train_data` - Dataset to train.\n- `roles` - Roles dict.\n- `verbose` - Controls the verbosity: the higher, the more messages.\n        <1  : messages are not displayed;\n        >=1 : the computation process for layers is displayed;\n        >=2 : the information about folds processing is also displayed;\n        >=3 : the hyperparameters optimization process is also displayed;\n        >=4 : the training process for every algorithm is displayed;\n\nNote: out-of-fold prediction is calculated during training and returned from the fit_predict method","metadata":{"papermill":{"duration":0.009424,"end_time":"2022-09-22T10:40:44.170889","exception":false,"start_time":"2022-09-22T10:40:44.161465","status":"completed"},"tags":[]}},{"cell_type":"code","source":"oof_pred_A = automl_A.fit_predict(train_data, roles=roles_A, verbose=3)\nprint(f'oof_pred:\\n{oof_pred_A}\\nShape = {oof_pred_A.shape}')\ngc.collect()","metadata":{"papermill":{"duration":720.991456,"end_time":"2022-09-22T10:52:45.171798","exception":false,"start_time":"2022-09-22T10:40:44.180342","status":"completed"},"tags":[],"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_pred_B = automl_B.fit_predict(train_data, roles=roles_B, verbose=3)\nprint(f'oof_pred:\\n{oof_pred_B}\\nShape = {oof_pred_B.shape}')\ngc.collect()","metadata":{"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oof_pred_train_A = oof_pred_A.data[:len(train_data), 0]\noof_pred_train_B = oof_pred_B.data[:len(train_data), 0]","metadata":{"papermill":{"duration":0.063172,"end_time":"2022-09-22T10:52:45.281375","exception":false,"start_time":"2022-09-22T10:52:45.218203","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfast_fi = automl_A.get_feature_scores('fast')\nfast_fi.set_index('Feature')['Importance'].plot.bar(figsize=(20, 10), grid=True)","metadata":{"papermill":{"duration":0.357738,"end_time":"2022-09-22T10:52:45.688469","exception":false,"start_time":"2022-09-22T10:52:45.330731","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfast_fi = automl_B.get_feature_scores('fast')\nfast_fi.set_index('Feature')['Importance'].plot.bar(figsize=(20, 10), grid=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Predict and save\nPredict and save submissions to .csv","metadata":{"papermill":{"duration":0.049738,"end_time":"2022-09-22T10:52:45.787574","exception":false,"start_time":"2022-09-22T10:52:45.737836","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\n\ntest_pred_A = automl_A.predict(test_data)\nprint(f'Prediction for test data:\\n{test_pred_A}\\nShape = {test_pred_A.shape}')","metadata":{"papermill":{"duration":1.830106,"end_time":"2022-09-22T10:52:47.667234","exception":false,"start_time":"2022-09-22T10:52:45.837128","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntest_pred_B = automl_B.predict(test_data)\nprint(f'Prediction for test data:\\n{test_pred_B}\\nShape = {test_pred_B.shape}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[TARGET_NAME_A] = test_pred_A.data[:, 0]\nsubmission[TARGET_NAME_B] = test_pred_B.data[:, 0]","metadata":{"papermill":{"duration":0.060514,"end_time":"2022-09-22T10:52:47.774178","exception":false,"start_time":"2022-09-22T10:52:47.713664","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('lightautoml.csv', index=False)","metadata":{"papermill":{"duration":0.086337,"end_time":"2022-09-22T10:52:47.907856","exception":false,"start_time":"2022-09-22T10:52:47.821519","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Additional materials","metadata":{"papermill":{"duration":0.046862,"end_time":"2022-09-22T10:52:48.001790","exception":false,"start_time":"2022-09-22T10:52:47.954928","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"- [Official LightAutoML github repo](https://github.com/AILab-MLTools/LightAutoML)\n- [LightAutoML documentation](https://lightautoml.readthedocs.io/en/latest)\n- [LightAutoML tutorials](https://github.com/AILab-MLTools/LightAutoML/tree/master/examples/tutorials)\n- LightAutoML course:\n    - [Part 1 - general overview](https://ods.ai/tracks/automl-course-part1) \n    - [Part 2 - LightAutoML specific applications](https://ods.ai/tracks/automl-course-part2)\n    - [Part 3 - LightAutoML customization](https://ods.ai/tracks/automl-course-part3)\n- [OpenDataScience AutoML benchmark leaderboard](https://ods.ai/competitions/automl-benchmark/leaderboard)","metadata":{"papermill":{"duration":0.047048,"end_time":"2022-09-22T10:52:48.096947","exception":false,"start_time":"2022-09-22T10:52:48.049899","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### If you still like the notebook, do not forget to put upvote for the notebook and the ⭐️ for github repo if you like it using the button below - one click for you, great pleasure for us ☺️","metadata":{"papermill":{"duration":0.046562,"end_time":"2022-09-22T10:52:48.190083","exception":false,"start_time":"2022-09-22T10:52:48.143521","status":"completed"},"tags":[]}},{"cell_type":"code","source":"s = '<iframe src=\"https://ghbtns.com/github-btn.html?user=sb-ai-lab&repo=LightAutoML&type=star&count=true&size=large\" frameborder=\"0\" scrolling=\"0\" width=\"170\" height=\"30\" title=\"LightAutoML GitHub\"></iframe>'\nHTML(s)","metadata":{"_kg_hide-input":true,"papermill":{"duration":0.061839,"end_time":"2022-09-22T10:52:48.299034","exception":false,"start_time":"2022-09-22T10:52:48.237195","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]}]}