{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T18:22:50.690389Z","iopub.execute_input":"2022-07-28T18:22:50.690874Z","iopub.status.idle":"2022-07-28T18:22:50.709714Z","shell.execute_reply.started":"2022-07-28T18:22:50.690834Z","shell.execute_reply":"2022-07-28T18:22:50.708287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 0. Task\n\nNeeds predict total sales for every product and store in the next month.","metadata":{}},{"cell_type":"markdown","source":"install LightAutoML and other libraries","metadata":{}},{"cell_type":"code","source":"%%capture\n!pip install lightautoml\n\n# QUICK WORKAROUND FOR PROBLEM WITH PANDAS\n!pip install -U pandas\n\n!pip install sweetviz","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:22:50.997032Z","iopub.execute_input":"2022-07-28T18:22:50.997924Z","iopub.status.idle":"2022-07-28T18:23:44.391423Z","shell.execute_reply.started":"2022-07-28T18:22:50.997870Z","shell.execute_reply":"2022-07-28T18:23:44.389462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import libraries","metadata":{}},{"cell_type":"code","source":"# Standard python libraries\nimport os\nimport time\n\n# Essential DS libraries\nimport numpy as np\nimport pandas as pd\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import train_test_split\nimport torch\n\n# LightAutoML presets, task and report generation\nfrom lightautoml.automl.presets.tabular_presets import TabularAutoML, TabularUtilizedAutoML\nfrom lightautoml.tasks import Task\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport sweetviz as sv\n\nimport re\n\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.feature_selection import f_classif # anova\nfrom sklearn.feature_selection import chi2 # хи-квадрат\n\nplt.style.use('ggplot')\npd.set_option('display.float_format', '{:.2f}'.format)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:44.394525Z","iopub.execute_input":"2022-07-28T18:23:44.394989Z","iopub.status.idle":"2022-07-28T18:23:44.419031Z","shell.execute_reply.started":"2022-07-28T18:23:44.394950Z","shell.execute_reply":"2022-07-28T18:23:44.417402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"LigthAutoML config","metadata":{}},{"cell_type":"code","source":"N_THREADS = 4\nN_FOLDS = 5\nRANDOM_STATE = 10\nTEST_SIZE = 0.2\nTIMEOUT = 30 * 60 # equal to 15 minutes\nTARGET_NAME = 'item_cnt_month'\n\nnp.random.seed(RANDOM_STATE)\ntorch.set_num_threads(N_THREADS)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:44.421257Z","iopub.execute_input":"2022-07-28T18:23:44.422652Z","iopub.status.idle":"2022-07-28T18:23:44.433209Z","shell.execute_reply.started":"2022-07-28T18:23:44.422593Z","shell.execute_reply":"2022-07-28T18:23:44.432083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Import data","metadata":{}},{"cell_type":"code","source":"INPUT_DIR = '../input/competitive-data-science-predict-future-sales/'","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:44.437970Z","iopub.execute_input":"2022-07-28T18:23:44.439082Z","iopub.status.idle":"2022-07-28T18:23:44.445578Z","shell.execute_reply.started":"2022-07-28T18:23:44.439027Z","shell.execute_reply":"2022-07-28T18:23:44.444536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"items = pd.read_csv(INPUT_DIR + 'items.csv')\ncategories = pd.read_csv(INPUT_DIR + 'item_categories.csv')\nshops = pd.read_csv(INPUT_DIR + 'shops.csv')\n\ndisplay(items.head(3))\ndisplay(categories.head(3))\ndisplay(shops.head(3))\n\ntrain_data = pd.read_csv(INPUT_DIR + 'sales_train.csv')\ntrain_data = train_data.merge(items, how='left', on='item_id')\n\ndisplay(train_data.head(3))\nprint(train_data.shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:44.448026Z","iopub.execute_input":"2022-07-28T18:23:44.448994Z","iopub.status.idle":"2022-07-28T18:23:47.171749Z","shell.execute_reply.started":"2022-07-28T18:23:44.448945Z","shell.execute_reply":"2022-07-28T18:23:47.170279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.merge(shops, how='left', on='shop_id')\ntrain_data = train_data.merge(categories, how='left', on='item_category_id')\n\ntrain_data.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:47.173610Z","iopub.execute_input":"2022-07-28T18:23:47.174650Z","iopub.status.idle":"2022-07-28T18:23:48.561790Z","shell.execute_reply.started":"2022-07-28T18:23:47.174573Z","shell.execute_reply":"2022-07-28T18:23:48.560536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(INPUT_DIR + 'test.csv')\ntest_data = test_data.join(items, how='left', on='item_id', rsuffix='_r')\n\ndisplay(test_data.head(3))\nprint(test_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:48.566185Z","iopub.execute_input":"2022-07-28T18:23:48.566637Z","iopub.status.idle":"2022-07-28T18:23:48.675838Z","shell.execute_reply.started":"2022-07-28T18:23:48.566603Z","shell.execute_reply":"2022-07-28T18:23:48.674171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Make test data like train data. We should add in test data next columns: item_cnt_month, month (next month) and item_price","metadata":{}},{"cell_type":"code","source":"train_data['date'] = pd.to_datetime(train_data['date'], format='%d.%m.%Y')\ntrain_data['month'] = train_data['date'].dt.month\ntrain_data['year'] = train_data['date'].dt.year","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:48.677481Z","iopub.execute_input":"2022-07-28T18:23:48.678514Z","iopub.status.idle":"2022-07-28T18:23:49.808525Z","shell.execute_reply.started":"2022-07-28T18:23:48.678477Z","shell.execute_reply":"2022-07-28T18:23:49.807079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"item_prices = train_data[['item_id', 'item_price']]\n\ntest_data = test_data.join(item_prices, how='left', on='item_id', rsuffix='_r')\n\n# next month\n# date_block_num - January 2013 is 0, February 2013 is 1,..., October 2015 is 33\ntest_data['date_block_num'] = train_data['date_block_num'].max() + 1\ntest_data['month'] = test_data['date_block_num'] % 12\ntest_data['year'] = train_data['year'].max()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:49.810788Z","iopub.execute_input":"2022-07-28T18:23:49.811167Z","iopub.status.idle":"2022-07-28T18:23:50.662340Z","shell.execute_reply.started":"2022-07-28T18:23:49.811130Z","shell.execute_reply":"2022-07-28T18:23:50.660733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = test_data.merge(shops, how='left', on='shop_id')\ntest_data = test_data.merge(categories, how='left', on='item_category_id')\n\ntest_data.set_index(test_data['ID'])\n\ntest_data.drop(['item_id_r', 'ID'], axis = 1, inplace=True)\ntest_data.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:50.669959Z","iopub.execute_input":"2022-07-28T18:23:50.670419Z","iopub.status.idle":"2022-07-28T18:23:50.886576Z","shell.execute_reply.started":"2022-07-28T18:23:50.670384Z","shell.execute_reply":"2022-07-28T18:23:50.885167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. Clear data\n\nLet's look up onto our dataframe","metadata":{}},{"cell_type":"code","source":"display(train_data.describe())\n\nhist = train_data.hist()\nhist","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:50.887888Z","iopub.execute_input":"2022-07-28T18:23:50.888240Z","iopub.status.idle":"2022-07-28T18:23:53.850928Z","shell.execute_reply.started":"2022-07-28T18:23:50.888208Z","shell.execute_reply":"2022-07-28T18:23:53.849251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- As we see, we have outliers by item_price and item_cnt_day, also item_cnt_day and item_price - can not be less 0, let's remove values less 0.\n- Also we can see that item_price and item_id has a very big std, thats mean we should make normalization","metadata":{}},{"cell_type":"code","source":"train_data = train_data[train_data['item_cnt_day'] >= 0]\ntrain_data = train_data[train_data['item_price'] > 0]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:53.853304Z","iopub.execute_input":"2022-07-28T18:23:53.853747Z","iopub.status.idle":"2022-07-28T18:23:54.597264Z","shell.execute_reply.started":"2022-07-28T18:23:53.853679Z","shell.execute_reply":"2022-07-28T18:23:54.595658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\nDropping duplicates","metadata":{}},{"cell_type":"code","source":"dupl_columns = list(train_data.columns)\nmask = train_data.duplicated(subset=dupl_columns)\nduplicates = train_data[mask]\nprint(f'Duplicates count: {duplicates.shape[0]}')\n\n# let's mark duplicates\ntrain_data = train_data.drop_duplicates(subset=dupl_columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:23:54.598836Z","iopub.execute_input":"2022-07-28T18:23:54.599188Z","iopub.status.idle":"2022-07-28T18:24:02.596014Z","shell.execute_reply.started":"2022-07-28T18:23:54.599153Z","shell.execute_reply":"2022-07-28T18:24:02.591786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_plots(data, column, title):\n    fig, axes = plt.subplots(nrows=1, ncols=1, figsize=(18, 8))\n    chart = sns.boxplot(data=data, x=column)\n    chart.set_title(title, fontsize=16)\n    chart.set_xlabel(column)\n\n    plt.tight_layout()\n    plt.show()\n    \nprint_plots(train_data, 'item_price', 'Boxplot price')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:02.602183Z","iopub.execute_input":"2022-07-28T18:24:02.604450Z","iopub.status.idle":"2022-07-28T18:24:03.830014Z","shell.execute_reply.started":"2022-07-28T18:24:02.604308Z","shell.execute_reply":"2022-07-28T18:24:03.828558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def outliers_iqr_mod(data, feature, left=1.5, right=1.5, log_scale=False):\n    if log_scale:\n        x = np.log(data[feature]+1)\n    else:\n        x = data[feature]\n        \n    quartile_1, quartile_3 = x.quantile(0.25), x.quantile(0.75),\n    iqr = quartile_3 - quartile_1\n    lower_bound = quartile_1 - (iqr * left)\n    upper_bound = quartile_3 + (iqr * right)\n    # removing outliers\n    outliers = data[(x < lower_bound) | (x > upper_bound)]\n    cleaned = data[(x > lower_bound) & (x < upper_bound)]\n    return outliers, cleaned\n\ndef add_outlier(data, column, title): \n    # by price\n    outliers_price, cleaned = outliers_iqr_mod(data, column, right=1.3)\n    print(f'irq price outliers: {outliers_price.shape[0]}')\n\n    if outliers_price.shape[0] > 0:\n        data.loc[outliers_price.index, 'outliers'] = True\n    \n    print_plots(cleaned, column, title)\n\n    print('Count of outliers: {}'.format(data[data['outliers'] == True]['outliers'].count()))\n\n    return cleaned","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:03.831779Z","iopub.execute_input":"2022-07-28T18:24:03.832906Z","iopub.status.idle":"2022-07-28T18:24:03.846782Z","shell.execute_reply.started":"2022-07-28T18:24:03.832865Z","shell.execute_reply":"2022-07-28T18:24:03.845074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.drop(['date'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:03.849765Z","iopub.execute_input":"2022-07-28T18:24:03.850919Z","iopub.status.idle":"2022-07-28T18:24:04.311518Z","shell.execute_reply.started":"2022-07-28T18:24:03.850876Z","shell.execute_reply":"2022-07-28T18:24:04.310176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = train_data.columns\n\nagg_arguments = {}\nfor col in columns:\n    if col == 'item_cnt_day':\n        agg_arguments[col] = 'sum'\n    #if col == 'item_price': \n    #    agg_arguments[col] = 'mean',     \n    elif col in ['date_block_num', 'item_id', 'shop_id']:\n        pass\n    else:\n        agg_arguments[col] = 'first'\n\ntrain_data = train_data.groupby(['item_id', 'date_block_num', 'shop_id']) \\\n.agg(agg_arguments) \\\n.reset_index() \\\n.rename(columns={'item_cnt_day': 'item_cnt_month'}) #, 'item_price': 'item_price_mean'})\n\ntrain_data.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:04.313908Z","iopub.execute_input":"2022-07-28T18:24:04.314357Z","iopub.status.idle":"2022-07-28T18:24:10.132292Z","shell.execute_reply.started":"2022-07-28T18:24:04.314310Z","shell.execute_reply":"2022-07-28T18:24:10.130558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:10.133746Z","iopub.execute_input":"2022-07-28T18:24:10.134114Z","iopub.status.idle":"2022-07-28T18:24:10.700426Z","shell.execute_reply.started":"2022-07-28T18:24:10.134081Z","shell.execute_reply":"2022-07-28T18:24:10.698771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ok, here we see that in data doesn't exist skips","metadata":{}},{"cell_type":"code","source":"train_shop_id = train_data['shop_id'].unique();\ntest_shop_id = test_data['shop_id'].unique();\n\ndisplay(train_shop_id, test_shop_id)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:10.702103Z","iopub.execute_input":"2022-07-28T18:24:10.702758Z","iopub.status.idle":"2022-07-28T18:24:10.726322Z","shell.execute_reply.started":"2022-07-28T18:24:10.702710Z","shell.execute_reply":"2022-07-28T18:24:10.724902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_shops = train_data[~train_data['shop_id'].isin(test_shop_id)]['shop_id'].unique()\ndrop_shops","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:10.728476Z","iopub.execute_input":"2022-07-28T18:24:10.729166Z","iopub.status.idle":"2022-07-28T18:24:10.798605Z","shell.execute_reply.started":"2022-07-28T18:24:10.729105Z","shell.execute_reply":"2022-07-28T18:24:10.797314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_categories = train_data['item_category_id'].unique();\ntest_categories = test_data['item_category_id'].unique();\n\ndisplay(train_categories, test_categories)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:10.800727Z","iopub.execute_input":"2022-07-28T18:24:10.801219Z","iopub.status.idle":"2022-07-28T18:24:10.828609Z","shell.execute_reply.started":"2022-07-28T18:24:10.801170Z","shell.execute_reply":"2022-07-28T18:24:10.827182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_cats = train_data[~train_data['item_category_id'].isin(test_categories)]['item_category_id'].unique()\ndrop_cats","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:10.830418Z","iopub.execute_input":"2022-07-28T18:24:10.831011Z","iopub.status.idle":"2022-07-28T18:24:10.855024Z","shell.execute_reply.started":"2022-07-28T18:24:10.830971Z","shell.execute_reply":"2022-07-28T18:24:10.853497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data[train_data['shop_id'].isin(drop_shops)]\ntrain_data = train_data[train_data['item_category_id'].isin(drop_cats)]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:10.856775Z","iopub.execute_input":"2022-07-28T18:24:10.857143Z","iopub.status.idle":"2022-07-28T18:24:10.937865Z","shell.execute_reply.started":"2022-07-28T18:24:10.857111Z","shell.execute_reply":"2022-07-28T18:24:10.936201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print_plots(train_data, 'item_cnt_month', 'Boxplot date_block_num')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:10.939780Z","iopub.execute_input":"2022-07-28T18:24:10.940137Z","iopub.status.idle":"2022-07-28T18:24:11.212014Z","shell.execute_reply.started":"2022-07-28T18:24:10.940105Z","shell.execute_reply":"2022-07-28T18:24:11.210749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check allocation of item_cnt_month","metadata":{}},{"cell_type":"code","source":"train_data = add_outlier(train_data, 'item_price', 'Boxplot price')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:11.213621Z","iopub.execute_input":"2022-07-28T18:24:11.214273Z","iopub.status.idle":"2022-07-28T18:24:11.994075Z","shell.execute_reply.started":"2022-07-28T18:24:11.214234Z","shell.execute_reply":"2022-07-28T18:24:11.992707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"as we see, item_cnt_month has many outlier items, let's remove it with seeking irq otlier method","metadata":{}},{"cell_type":"code","source":"train_data = add_outlier(train_data, 'item_cnt_month', 'Boxplot cnt month')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:11.995851Z","iopub.execute_input":"2022-07-28T18:24:11.997039Z","iopub.status.idle":"2022-07-28T18:24:12.311996Z","shell.execute_reply.started":"2022-07-28T18:24:11.996998Z","shell.execute_reply":"2022-07-28T18:24:12.310430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. EDA","metadata":{}},{"cell_type":"markdown","source":"Let's make fast eda with [library pandas-profiling](https://pandas-profiling.ydata.ai/docs/master/index.html)","metadata":{}},{"cell_type":"code","source":"report = sv.analyze(train_data)\nreport.show_notebook()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:12.313853Z","iopub.execute_input":"2022-07-28T18:24:12.314218Z","iopub.status.idle":"2022-07-28T18:24:20.075352Z","shell.execute_reply.started":"2022-07-28T18:24:12.314184Z","shell.execute_reply":"2022-07-28T18:24:20.074044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets select some new signs from \n - category_name: we can select few categories\n - shop_name we can select city\n - item_name we can select carrier type","metadata":{}},{"cell_type":"code","source":"train_data['item_category_name'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:20.086347Z","iopub.execute_input":"2022-07-28T18:24:20.086822Z","iopub.status.idle":"2022-07-28T18:24:20.097552Z","shell.execute_reply.started":"2022-07-28T18:24:20.086784Z","shell.execute_reply":"2022-07-28T18:24:20.095668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['shop_name'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:20.099419Z","iopub.execute_input":"2022-07-28T18:24:20.100639Z","iopub.status.idle":"2022-07-28T18:24:20.110789Z","shell.execute_reply.started":"2022-07-28T18:24:20.100600Z","shell.execute_reply":"2022-07-28T18:24:20.108927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_cat(item):\n    cats = item.split(' - ')\n    \n    return cats[0]\n    \ndef get_subcat(item):\n    cats = item.split(' - ')\n    \n    if 'Билеты (Цифра)' == cats[0]:\n        return 'Билеты'\n    \n    if len(cats) < 2: \n        return 'Прочие'\n        \n    if 'Цифра' in cats[1]:\n        return 'digit'\n    \n    if 'Live!' in cats[1]:\n        return 'Прочие'\n    \n    return cats[1]\n\ntrain_data['category'] = train_data['item_category_name'].apply(get_cat)\ntrain_data['subcategory'] = train_data['item_category_name'].apply(get_subcat)\n\ntest_data['category'] = test_data['item_category_name'].apply(get_cat)\ntest_data['subcategory'] = test_data['item_category_name'].apply(get_subcat)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:20.113003Z","iopub.execute_input":"2022-07-28T18:24:20.114435Z","iopub.status.idle":"2022-07-28T18:24:20.562396Z","shell.execute_reply.started":"2022-07-28T18:24:20.114385Z","shell.execute_reply":"2022-07-28T18:24:20.560999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_city(item):\n    if 'Цифровой' in item or 'Интернет-магазин' in item:\n        return 'digital'\n    \n    name = item.split(' ')\n    name = re.sub(r'(?u)[^\\w]+', '', name[0])\n    \n    if name == 'РостовНаДону':\n        return 'Ростов-на-Дону'\n    if name == 'СПб':\n        return 'Санкт-Петербург'\n    if name == 'ННовгород':\n        return 'Нижний Новгород'\n    if name == 'Сергиев':\n        return 'Сергиев Посад'\n    \n    return name\n    \ntrain_data['city'] = train_data['shop_name'].apply(get_city)\ntest_data['city'] = test_data['shop_name'].apply(get_city)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:20.564242Z","iopub.execute_input":"2022-07-28T18:24:20.564652Z","iopub.status.idle":"2022-07-28T18:24:21.317327Z","shell.execute_reply.started":"2022-07-28T18:24:20.564617Z","shell.execute_reply":"2022-07-28T18:24:21.315766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"city_info = pd.read_csv(INPUT_DIR + '../cityinfo/data.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:21.319330Z","iopub.execute_input":"2022-07-28T18:24:21.319757Z","iopub.status.idle":"2022-07-28T18:24:22.172150Z","shell.execute_reply.started":"2022-07-28T18:24:21.319719Z","shell.execute_reply":"2022-07-28T18:24:22.170566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data.shape[0], test_data.shape[0])\ntest_data['city'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:22.174314Z","iopub.execute_input":"2022-07-28T18:24:22.174778Z","iopub.status.idle":"2022-07-28T18:24:22.230680Z","shell.execute_reply.started":"2022-07-28T18:24:22.174736Z","shell.execute_reply":"2022-07-28T18:24:22.228923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cities = set(list(test_data['city'].unique()) + list(train_data['city'].unique()))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:22.232016Z","iopub.execute_input":"2022-07-28T18:24:22.232401Z","iopub.status.idle":"2022-07-28T18:24:22.269779Z","shell.execute_reply.started":"2022-07-28T18:24:22.232368Z","shell.execute_reply":"2022-07-28T18:24:22.268586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_city_info(sn):\n    \n    if sn == 'Уфа':\n        conditions = (city_info['municipality'] == sn) & (city_info['type'] == 'р-н')\n    elif sn == 'Адыгея':\n        conditions = (city_info['region'] == 'Республика Адыгея')\n    else:\n        conditions = (city_info['settlement'] == sn)\n\n    cities_info = city_info[conditions]\n    population = cities_info.groupby('region').agg({'population': 'sum'}).reset_index()['population'].max()\n    \n    return population\n    \ncities_sizes = {}\nfor city in cities:\n    cities_sizes[city] = get_city_info(city)\n    \ncities_sizes","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:22.271832Z","iopub.execute_input":"2022-07-28T18:24:22.273049Z","iopub.status.idle":"2022-07-28T18:24:23.286942Z","shell.execute_reply.started":"2022-07-28T18:24:22.272990Z","shell.execute_reply":"2022-07-28T18:24:23.285325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_city_info(data):\n    data['city_size'] = data['city'].apply(lambda x: cities_sizes[x])\n    mean_size = data[(data['city'] != 'Москва') & (data['city'] != 'Санкт-Петербург')]['city_size'].mean()\n    data['city_size'] = data['city_size'].fillna(mean_size)\n    return data\n\ntrain_data = set_city_info(train_data)\ntest_data = set_city_info(test_data)\n\ntrain_data.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:23.288927Z","iopub.execute_input":"2022-07-28T18:24:23.289385Z","iopub.status.idle":"2022-07-28T18:24:23.639527Z","shell.execute_reply.started":"2022-07-28T18:24:23.289346Z","shell.execute_reply":"2022-07-28T18:24:23.637997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def avg_chart(data, column, title, dest_column = 'item_cnt_month'):\n    pivotdata = data[[column, dest_column]].groupby(column).mean()\n    fig5, ax5 = plt.subplots(figsize=(15, 5))\n    plt.suptitle(title, size=16)\n    bar_pivot = sns.barplot(\n        x=pivotdata.index, \n        y=pivotdata[dest_column])\n\n    for p in bar_pivot.patches:\n        bar_pivot.annotate('{:.2f}'.format(p.get_height()), \n                           (p.get_x()+0.4, p.get_height()),\n                           ha='center', va='bottom', fontsize=14)\n\n\navg_chart(train_data, 'city', 'Avg by city')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:23.641257Z","iopub.execute_input":"2022-07-28T18:24:23.641611Z","iopub.status.idle":"2022-07-28T18:24:24.004040Z","shell.execute_reply.started":"2022-07-28T18:24:23.641578Z","shell.execute_reply":"2022-07-28T18:24:24.002237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_chart(train_data, 'city', 'City cize', 'city_size')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:24.005989Z","iopub.execute_input":"2022-07-28T18:24:24.006393Z","iopub.status.idle":"2022-07-28T18:24:24.369769Z","shell.execute_reply.started":"2022-07-28T18:24:24.006359Z","shell.execute_reply":"2022-07-28T18:24:24.368059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_chart(train_data, 'category', 'Avg by category')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:24.372490Z","iopub.execute_input":"2022-07-28T18:24:24.373065Z","iopub.status.idle":"2022-07-28T18:24:24.759485Z","shell.execute_reply.started":"2022-07-28T18:24:24.373013Z","shell.execute_reply":"2022-07-28T18:24:24.758393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_chart(train_data, 'subcategory', 'Avg by sub category')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:24.761415Z","iopub.execute_input":"2022-07-28T18:24:24.762233Z","iopub.status.idle":"2022-07-28T18:24:25.085743Z","shell.execute_reply.started":"2022-07-28T18:24:24.762180Z","shell.execute_reply":"2022-07-28T18:24:25.084380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_chart(train_data, 'date_block_num', 'Avg by year month')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:25.087478Z","iopub.execute_input":"2022-07-28T18:24:25.087905Z","iopub.status.idle":"2022-07-28T18:24:25.720892Z","shell.execute_reply.started":"2022-07-28T18:24:25.087867Z","shell.execute_reply":"2022-07-28T18:24:25.719586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_chart(train_data, 'month', 'Avg by month')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:25.723169Z","iopub.execute_input":"2022-07-28T18:24:25.724005Z","iopub.status.idle":"2022-07-28T18:24:26.099623Z","shell.execute_reply.started":"2022-07-28T18:24:25.723952Z","shell.execute_reply":"2022-07-28T18:24:26.098300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"avg_chart(train_data, 'year', 'Avg by year')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:26.102898Z","iopub.execute_input":"2022-07-28T18:24:26.103343Z","iopub.status.idle":"2022-07-28T18:24:26.353762Z","shell.execute_reply.started":"2022-07-28T18:24:26.103308Z","shell.execute_reply":"2022-07-28T18:24:26.352044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"lets encode signs","metadata":{}},{"cell_type":"code","source":"import category_encoders as ce\n\ndef bin_encoding(data, column):\n    encoder = ce.BinaryEncoder(cols=[column])\n    encoded = encoder.fit_transform(data[column])\n    return pd.concat([data, encoded], axis=1)\n    \ndef ord_encoding(data, column):\n    ord_encoder = ce.OrdinalEncoder(cols=[column])\n    data[column + '_cat'] = ord_encoder.fit_transform(data[column])\n    return data\n\ntrain_data = bin_encoding(train_data, 'category')\ntest_data = bin_encoding(test_data, 'category')\n\ntrain_data = bin_encoding(train_data, 'subcategory')\ntest_data = bin_encoding(test_data, 'subcategory')\n\ntrain_data = bin_encoding(train_data, 'city')\ntest_data = bin_encoding(test_data, 'city')\n\ntrain_data = ord_encoding(train_data, 'item_name')\ntest_data = ord_encoding(test_data, 'item_name')\n\ndrop_columns = ['category', 'subcategory', 'item_category_name', 'city', 'item_name', 'shop_name'];\ntest_data.drop(drop_columns, axis = 1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:26.355440Z","iopub.execute_input":"2022-07-28T18:24:26.355847Z","iopub.status.idle":"2022-07-28T18:24:28.455031Z","shell.execute_reply.started":"2022-07-28T18:24:26.355812Z","shell.execute_reply":"2022-07-28T18:24:28.453111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_item_lags(data, feature_name, nlags=3):\n    df_tmp = data[['date_block_num', 'shop_id', 'item_id', feature_name]]\n    \n    for i in range(nlags, 0, -1):\n\n        lag_feature_name = feature_name +'_lag-' + str(i)\n        \n        df_shifted = df_tmp.copy()\n        df_shifted.columns = ['date_block_num', 'shop_id', 'item_id', lag_feature_name]\n        df_shifted['date_block_num'] += i\n        data = pd.merge(data, df_shifted, on=['date_block_num', 'shop_id', 'item_id'], how='left')\n        \n        data[lag_feature_name] = data[lag_feature_name].fillna(0).astype('float32')\n\n    del df_tmp\n\n    return data\n\ndef add_shop_lags(data, feature_name, nlags=3, dnc=False):\n\n    mean_feature_name = feature_name + '_mean'\n\n    df_tmp = data[['date_block_num', 'shop_id', feature_name, mean_feature_name]]\n\n    for i in range(nlags, 0, -1):\n\n        lag_feature_name = mean_feature_name + '_lag-' + str(i)\n\n        df_shifted = df_tmp.copy()\n        df_shifted.columns = ['date_block_num', 'shop_id', feature_name, lag_feature_name]\n        df_shifted['date_block_num'] += i\n        data = pd.merge(data, df_shifted, on=['date_block_num', 'shop_id', feature_name], how='left')\n        data[lag_feature_name] = data[lag_feature_name].fillna(0).astype('float32')\n\n    del df_tmp\n    del df_shifted\n\n    return data\n    \n    \ndef add_lags(data):\n    item_mean_features = ['city'], \n    shop_mean_features = ['item_category_id'], \n\n    # Critical point: True or False (Affects qmean calculation)\n    data = add_item_lags(data, 'item_cnt_month')\n    data = add_item_lags(data, 'item_price')\n\n    for mf in item_mean_features:\n        data = add_item_lags(data, mf)\n\n    for mf in shop_mean_features:\n        data = add_shop_lags(data, mf)\n    \n    return data\n    \n#add_lags(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:28.456852Z","iopub.execute_input":"2022-07-28T18:24:28.457228Z","iopub.status.idle":"2022-07-28T18:24:28.481629Z","shell.execute_reply.started":"2022-07-28T18:24:28.457195Z","shell.execute_reply":"2022-07-28T18:24:28.474815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.drop(['category', 'subcategory', 'item_category_name', 'city', 'item_name', 'shop_name'], axis = 1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:28.483803Z","iopub.execute_input":"2022-07-28T18:24:28.488990Z","iopub.status.idle":"2022-07-28T18:24:28.503487Z","shell.execute_reply.started":"2022-07-28T18:24:28.488907Z","shell.execute_reply":"2022-07-28T18:24:28.502226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:28.506383Z","iopub.execute_input":"2022-07-28T18:24:28.507578Z","iopub.status.idle":"2022-07-28T18:24:28.518973Z","shell.execute_reply.started":"2022-07-28T18:24:28.507528Z","shell.execute_reply":"2022-07-28T18:24:28.517784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Normalizing","metadata":{}},{"cell_type":"code","source":"num_columns = [\n    'item_price', \n    'item_id',\n    'city_size'\n]\n\ndef normal_chart(data, columns):\n    fig = plt.figure(figsize=(10, 8))\n\n    for col in columns:\n        kde = sns.kdeplot(data[col], label =col)\n\n    kde.set_title('Исходные распределения')\n    plt.legend();\n\nnormal_chart(train_data, num_columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:28.521128Z","iopub.execute_input":"2022-07-28T18:24:28.522221Z","iopub.status.idle":"2022-07-28T18:24:28.920991Z","shell.execute_reply.started":"2022-07-28T18:24:28.522168Z","shell.execute_reply":"2022-07-28T18:24:28.919782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:28.923197Z","iopub.execute_input":"2022-07-28T18:24:28.924356Z","iopub.status.idle":"2022-07-28T18:24:28.955145Z","shell.execute_reply.started":"2022-07-28T18:24:28.924300Z","shell.execute_reply":"2022-07-28T18:24:28.953739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom sklearn.preprocessing import RobustScaler\n\ndef normalizing(data, columns):\n    \n    trans = MinMaxScaler()\n    for col in columns:\n        df = pd.DataFrame(data[col])\n        data[col + '_norm'] = trans.fit_transform(df)\n        data[col + '_norm'] = data[col + '_norm'].fillna(0)\n        \n    return data\n\ntrain_data = normalizing(train_data, ['item_id', 'item_price', 'city_size'])\ntest_data = normalizing(test_data, ['item_id', 'item_price', 'city_size'])","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:28.957395Z","iopub.execute_input":"2022-07-28T18:24:28.957930Z","iopub.status.idle":"2022-07-28T18:24:29.008219Z","shell.execute_reply.started":"2022-07-28T18:24:28.957868Z","shell.execute_reply":"2022-07-28T18:24:29.006873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normal_chart(train_data, ['item_price_norm', 'item_id_norm', 'city_size_norm'])\nnormal_chart(test_data, ['item_price_norm', 'item_id_norm', 'city_size_norm'])","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:29.012653Z","iopub.execute_input":"2022-07-28T18:24:29.013128Z","iopub.status.idle":"2022-07-28T18:24:31.878182Z","shell.execute_reply.started":"2022-07-28T18:24:29.013088Z","shell.execute_reply":"2022-07-28T18:24:31.876898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:31.880013Z","iopub.execute_input":"2022-07-28T18:24:31.881262Z","iopub.status.idle":"2022-07-28T18:24:31.968839Z","shell.execute_reply.started":"2022-07-28T18:24:31.881208Z","shell.execute_reply":"2022-07-28T18:24:31.967191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Multicollinearity analysis","metadata":{}},{"cell_type":"code","source":"train_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:31.971105Z","iopub.execute_input":"2022-07-28T18:24:31.971673Z","iopub.status.idle":"2022-07-28T18:24:31.981153Z","shell.execute_reply.started":"2022-07-28T18:24:31.971621Z","shell.execute_reply":"2022-07-28T18:24:31.979891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nat_cols = ['item_price_norm', 'item_id_norm', 'city_size_norm']\n\n# categrorial signs\ncat_cols = ['date_block_num', 'shop_id', 'item_cnt_month',\n       'month', 'year', 'category_0', 'category_1', 'category_2', 'category_3',\n       'subcategory_0', 'subcategory_1', 'subcategory_2', 'subcategory_3', 'city_0', 'city_1',\n       'city_2', 'city_3', 'item_name_cat']","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:31.983460Z","iopub.execute_input":"2022-07-28T18:24:31.984238Z","iopub.status.idle":"2022-07-28T18:24:31.993340Z","shell.execute_reply.started":"2022-07-28T18:24:31.984183Z","shell.execute_reply":"2022-07-28T18:24:31.992205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correlation_chart(data, columns, method='pearson', title='Correlation'):\n    fig_, ax_ = plt.subplots(figsize=(15, 12))\n    corr = data[columns].corr(method=method)\n    mask = np.triu(np.ones_like(corr, dtype=bool))\n\n    sns.heatmap(corr, \n                annot=True, \n                linewidths=0.1, \n                ax=ax_, \n                mask=mask, \n                cmap='viridis',\n                fmt='.1g')\n    ax_.set_title(title, fontsize=18)\n    plt.show()\n\ncorrelation_chart(train_data, cat_cols, 'spearman', 'Correlation of categorial features')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:31.995555Z","iopub.execute_input":"2022-07-28T18:24:31.997260Z","iopub.status.idle":"2022-07-28T18:24:33.139449Z","shell.execute_reply.started":"2022-07-28T18:24:31.997207Z","shell.execute_reply":"2022-07-28T18:24:33.137795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_chart(train_data, nat_cols, 'pearson', 'Correlation of categorial features')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:33.141876Z","iopub.execute_input":"2022-07-28T18:24:33.142304Z","iopub.status.idle":"2022-07-28T18:24:33.459078Z","shell.execute_reply.started":"2022-07-28T18:24:33.142268Z","shell.execute_reply":"2022-07-28T18:24:33.457523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"we can remove year column, becase of it has correlation more than 0.7","metadata":{}},{"cell_type":"code","source":"drop_columns = ['item_id', 'item_price', 'item_category_id', 'city_size',\n                'subcategory_1', 'subcategory_2', 'subcategory_3', 'city_3']\n\ntest_data.drop(drop_columns, axis = 1, inplace=True)\ntrain_data.drop(drop_columns, axis = 1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:33.460923Z","iopub.execute_input":"2022-07-28T18:24:33.461303Z","iopub.status.idle":"2022-07-28T18:24:33.487100Z","shell.execute_reply.started":"2022-07-28T18:24:33.461270Z","shell.execute_reply":"2022-07-28T18:24:33.485511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ASSESSMENT OF THE SIGNIFICANCE OF FEATURES","metadata":{}},{"cell_type":"code","source":"X = train_data.drop(['item_cnt_month'], axis = 1)  \ny = train_data['item_cnt_month']","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:33.490089Z","iopub.execute_input":"2022-07-28T18:24:33.491107Z","iopub.status.idle":"2022-07-28T18:24:33.498670Z","shell.execute_reply.started":"2022-07-28T18:24:33.491052Z","shell.execute_reply":"2022-07-28T18:24:33.497773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:33.500964Z","iopub.execute_input":"2022-07-28T18:24:33.501532Z","iopub.status.idle":"2022-07-28T18:24:33.525928Z","shell.execute_reply.started":"2022-07-28T18:24:33.501489Z","shell.execute_reply":"2022-07-28T18:24:33.524127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=y.astype('int')\n\nimp_num = pd.Series(f_classif(X[X.columns], y)[0], index = X.columns)\nimp_num.sort_values(inplace = True)\n\nfig5, ax5 = plt.subplots(figsize=(15, 20))\nimp_num.plot(kind = 'barh')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:33.529019Z","iopub.execute_input":"2022-07-28T18:24:33.529842Z","iopub.status.idle":"2022-07-28T18:24:33.922465Z","shell.execute_reply.started":"2022-07-28T18:24:33.529796Z","shell.execute_reply":"2022-07-28T18:24:33.921449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Making model","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv(INPUT_DIR + 'sample_submission.csv')\nprint(submission.shape)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:33.923910Z","iopub.execute_input":"2022-07-28T18:24:33.924276Z","iopub.status.idle":"2022-07-28T18:24:33.993242Z","shell.execute_reply.started":"2022-07-28T18:24:33.924241Z","shell.execute_reply":"2022-07-28T18:24:33.991953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_data, te_data = train_test_split(\n    train_data, \n    test_size=TEST_SIZE, \n    random_state=RANDOM_STATE\n)\n\nprint(f'Data splitted. Parts sizes: tr_data = {tr_data.shape}, te_data = {te_data.shape}')\n\ntr_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:33.995195Z","iopub.execute_input":"2022-07-28T18:24:33.995978Z","iopub.status.idle":"2022-07-28T18:24:34.023067Z","shell.execute_reply.started":"2022-07-28T18:24:33.995930Z","shell.execute_reply":"2022-07-28T18:24:34.021763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Task defenition\ntask = Task('reg', loss = 'mae', metric = 'mae')\n# Roles setup\nroles = {\n    'target': TARGET_NAME,\n}\n#TabularUtilizedAutoML\nautoml = TabularUtilizedAutoML(\n    task = task, \n    timeout = TIMEOUT,\n    cpu_limit = N_THREADS,\n    reader_params = {'n_jobs': N_THREADS, 'cv': N_FOLDS, 'random_state': RANDOM_STATE}\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:34.025014Z","iopub.execute_input":"2022-07-28T18:24:34.025815Z","iopub.status.idle":"2022-07-28T18:24:34.060898Z","shell.execute_reply.started":"2022-07-28T18:24:34.025765Z","shell.execute_reply":"2022-07-28T18:24:34.059375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"AutoML trainig","metadata":{}},{"cell_type":"code","source":"%%time\noof_pred = automl.fit_predict(tr_data, roles = roles, verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:24:34.063471Z","iopub.execute_input":"2022-07-28T18:24:34.064007Z","iopub.status.idle":"2022-07-28T18:52:24.406013Z","shell.execute_reply.started":"2022-07-28T18:24:34.063958Z","shell.execute_reply":"2022-07-28T18:52:24.404262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(automl.create_model_str_desc())","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:52:24.408356Z","iopub.execute_input":"2022-07-28T18:52:24.408997Z","iopub.status.idle":"2022-07-28T18:52:24.417310Z","shell.execute_reply.started":"2022-07-28T18:52:24.408950Z","shell.execute_reply":"2022-07-28T18:52:24.415782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Predict","metadata":{}},{"cell_type":"code","source":"%%time\nte_pred = automl.predict(te_data)\nprint(f'Prediction for te_data:\\n{te_pred}\\nShape = {te_pred.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:52:24.419304Z","iopub.execute_input":"2022-07-28T18:52:24.419674Z","iopub.status.idle":"2022-07-28T18:52:25.322202Z","shell.execute_reply.started":"2022-07-28T18:52:24.419641Z","shell.execute_reply":"2022-07-28T18:52:25.320412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'TRAIN out-of-fold score: {mean_absolute_error(tr_data[TARGET_NAME].values, oof_pred.data[:, 0])}')\nprint(f'HOLDOUT score: {mean_absolute_error(te_data[TARGET_NAME].values, te_pred.data[:, 0])}')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:52:25.324637Z","iopub.execute_input":"2022-07-28T18:52:25.325022Z","iopub.status.idle":"2022-07-28T18:52:25.343936Z","shell.execute_reply.started":"2022-07-28T18:52:25.324988Z","shell.execute_reply":"2022-07-28T18:52:25.342262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Feature importances calculation","metadata":{}},{"cell_type":"markdown","source":"\nFor feature importances calculation we have 2 different methods in LightAutoML:\n\nFast (fast) - this method uses feature importances from feature selector LGBM model inside LightAutoML. It works extremely fast and almost always (almost because of situations, when feature selection is turned off or selector was removed from the final models with all GBM models). no need to use new labelled data.\nAccurate (accurate) - this method calculate features permutation importances for the whole LightAutoML model based on the new labelled data. It always works but can take a lot of time to finish (depending on the model structure, new labelled dataset size etc.).\n","metadata":{}},{"cell_type":"code","source":"%%time\n\n# Fast feature importances calculation\nfast_fi = automl.get_feature_scores('fast')\nfast_fi.set_index('Feature')['Importance'].plot.bar(figsize = (30, 10), grid = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:52:25.345973Z","iopub.execute_input":"2022-07-28T18:52:25.346484Z","iopub.status.idle":"2022-07-28T18:52:25.777629Z","shell.execute_reply.started":"2022-07-28T18:52:25.346436Z","shell.execute_reply":"2022-07-28T18:52:25.776216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\n# Accurate feature importances calculation (Permutation importances) -  can take long time to calculate on bigger datasets\naccurate_fi = automl.get_feature_scores('accurate', te_data, silent = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:52:25.779541Z","iopub.execute_input":"2022-07-28T18:52:25.782335Z","iopub.status.idle":"2022-07-28T18:52:38.053831Z","shell.execute_reply.started":"2022-07-28T18:52:25.782266Z","shell.execute_reply":"2022-07-28T18:52:38.051972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accurate_fi.set_index('Feature')['Importance'].plot.bar(figsize = (30, 10), grid = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:52:38.056112Z","iopub.execute_input":"2022-07-28T18:52:38.056677Z","iopub.status.idle":"2022-07-28T18:52:38.483375Z","shell.execute_reply.started":"2022-07-28T18:52:38.056622Z","shell.execute_reply":"2022-07-28T18:52:38.482086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Predict for test dataset","metadata":{}},{"cell_type":"code","source":"test_pred = automl.predict(test_data)\nprint(f'Prediction for te_data:\\n{test_pred}\\nShape = {test_pred.shape}')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:52:38.485187Z","iopub.execute_input":"2022-07-28T18:52:38.485558Z","iopub.status.idle":"2022-07-28T18:54:27.451712Z","shell.execute_reply.started":"2022-07-28T18:52:38.485525Z","shell.execute_reply":"2022-07-28T18:54:27.449866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission[TARGET_NAME] = test_pred.data[:, 0]\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:54:27.453738Z","iopub.execute_input":"2022-07-28T18:54:27.454256Z","iopub.status.idle":"2022-07-28T18:54:27.474114Z","shell.execute_reply.started":"2022-07-28T18:54:27.454216Z","shell.execute_reply":"2022-07-28T18:54:27.472538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('final.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T18:54:27.477138Z","iopub.execute_input":"2022-07-28T18:54:27.477634Z","iopub.status.idle":"2022-07-28T18:54:28.102067Z","shell.execute_reply.started":"2022-07-28T18:54:27.477599Z","shell.execute_reply":"2022-07-28T18:54:28.100429Z"},"trusted":true},"execution_count":null,"outputs":[]}]}