{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":8586,"databundleVersionId":868729,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:33:00.524425Z","iopub.execute_input":"2024-11-11T12:33:00.525201Z","iopub.status.idle":"2024-11-11T12:33:00.532689Z","shell.execute_reply.started":"2024-11-11T12:33:00.525161Z","shell.execute_reply":"2024-11-11T12:33:00.531672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport pandas as pd","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:35:34.678533Z","iopub.execute_input":"2024-11-11T12:35:34.67939Z","iopub.status.idle":"2024-11-11T12:35:34.885844Z","shell.execute_reply.started":"2024-11-11T12:35:34.67935Z","shell.execute_reply":"2024-11-11T12:35:34.884927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/avito-demand-prediction/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/avito-demand-prediction/test.csv\")\n\ndf_train.shape, df_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:09:31.457178Z","iopub.execute_input":"2024-11-11T11:09:31.458544Z","iopub.status.idle":"2024-11-11T11:10:18.923922Z","shell.execute_reply.started":"2024-11-11T11:09:31.45848Z","shell.execute_reply":"2024-11-11T11:10:18.922671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:11:00.242095Z","iopub.execute_input":"2024-11-11T11:11:00.243659Z","iopub.status.idle":"2024-11-11T11:11:02.554776Z","shell.execute_reply.started":"2024-11-11T11:11:00.243533Z","shell.execute_reply":"2024-11-11T11:11:02.553256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:11:02.556852Z","iopub.execute_input":"2024-11-11T11:11:02.557254Z","iopub.status.idle":"2024-11-11T11:11:02.587765Z","shell.execute_reply.started":"2024-11-11T11:11:02.557212Z","shell.execute_reply":"2024-11-11T11:11:02.586422Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Анализ признаков\nПока будем проводить анализ без учета изображений.","metadata":{}},{"cell_type":"code","source":"df_train = df_train.drop(['image', 'image_top_1'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:11:09.106553Z","iopub.execute_input":"2024-11-11T11:11:09.107141Z","iopub.status.idle":"2024-11-11T11:11:09.487434Z","shell.execute_reply.started":"2024-11-11T11:11:09.107087Z","shell.execute_reply":"2024-11-11T11:11:09.486154Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Как было видно выше из df_train.info(), в данных присутствуют пропущенные значения. Посмотрим, в каких колонках они содердатся и в каком количестве","metadata":{}},{"cell_type":"code","source":"nan_summary = pd.DataFrame({\n    'NaN_Count': df_train.isna().sum(),\n    'NaN_Ratio': df_train.isna().mean()\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:11:13.147336Z","iopub.execute_input":"2024-11-11T11:11:13.148652Z","iopub.status.idle":"2024-11-11T11:11:17.388874Z","shell.execute_reply.started":"2024-11-11T11:11:13.148561Z","shell.execute_reply":"2024-11-11T11:11:17.387429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nan_summary[nan_summary['NaN_Count'] > 0].sort_values('NaN_Ratio', ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:11:21.338004Z","iopub.execute_input":"2024-11-11T11:11:21.338669Z","iopub.status.idle":"2024-11-11T11:11:21.359444Z","shell.execute_reply.started":"2024-11-11T11:11:21.338575Z","shell.execute_reply":"2024-11-11T11:11:21.357737Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Как мы видим, пропущенные значения содержаться в, видимо, необязательных для заполнения карточки товара полях, а именно, параметрах, которые уточняют категорию товара, в описании. Пропущенные значения цены, видимо, говорят о договорной цене.   \nДалее изучим внимательнее уникальные значения признаков и целевой переменной (вероятности сделки).","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 5))\n\ncols = df_train.columns\nuniques = [len(df_train[col].unique()) for col in cols]\n\nax = sns.barplot(x=cols, y=uniques, palette='BrBG', log=True)\nax.set(xlabel='Признак', ylabel='log(unique count)', title='Количество уникальных значений')\n\n\nfor p, uniq in zip(ax.patches, uniques):\n    ax.text(p.get_x() + p.get_width()/2.,\n            uniq + 10,\n            uniq,\n            ha=\"center\") \n\nax.set_xticklabels(ax.get_xticklabels(), rotation=45);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:11:25.947492Z","iopub.execute_input":"2024-11-11T11:11:25.948093Z","iopub.status.idle":"2024-11-11T11:11:31.592088Z","shell.execute_reply.started":"2024-11-11T11:11:25.948044Z","shell.execute_reply":"2024-11-11T11:11:31.590365Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Видим, что в user_id уникальных значений меньше, чем в item_id, что логично, ведь один пользователь мог разместить несколько объявлений. Так же много уникальных значений в поле description, тоже логично, ведь, скорее всего, почти каждое описание уникально, за редкими исключениями и NaN'ами.\nРассмотрим подробнее каждый признак и целевую переменную.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.histplot(df_train['deal_probability'], kde=True, bins=20, color='skyblue')\nplt.title(\"Распределение вероятности для Deal probability\")\nplt.xlabel(\"Deal_probability\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:11:41.074507Z","iopub.execute_input":"2024-11-11T11:11:41.075106Z","iopub.status.idle":"2024-11-11T11:11:49.380725Z","shell.execute_reply.started":"2024-11-11T11:11:41.075051Z","shell.execute_reply":"2024-11-11T11:11:49.379299Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Видим, что для большинства объявлений вероятность совершения сделки близка к нулю, и лишь для небольшого числа объектов она больше 0.8. Пока для удобства переведем вероятность в категориальную переменную.","metadata":{}},{"cell_type":"code","source":"df_train[\"deal_prob_cat\"] = pd.cut(df_train.deal_probability, bins=10)\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:11:54.122944Z","iopub.execute_input":"2024-11-11T11:11:54.123515Z","iopub.status.idle":"2024-11-11T11:11:54.214657Z","shell.execute_reply.started":"2024-11-11T11:11:54.123462Z","shell.execute_reply":"2024-11-11T11:11:54.213108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature = 'region' # Выбор признака\ncategory_counts = df_train[feature].value_counts()\n#category_counts.index = category_counts.index.str.split().str[0]\n\nplt.figure(figsize=(12, 8))\nsns.barplot(x=category_counts.index, y=category_counts.values, palette='BrBG')\nplt.title(f\"Количество объявлений для категории {feature}\")\nplt.xlabel(\"Категории\")\nplt.ylabel(\"Количество объявлений\")\nplt.xticks(rotation=45, ha=\"right\")  \n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:11:58.00713Z","iopub.execute_input":"2024-11-11T11:11:58.007752Z","iopub.status.idle":"2024-11-11T11:11:58.976101Z","shell.execute_reply.started":"2024-11-11T11:11:58.007695Z","shell.execute_reply":"2024-11-11T11:11:58.974682Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Небольшая сводка:  \n1. По признаку region наибольшое количество объявлений в субъектах: Краснодарский край, Свердловская обл., Ростовская обл., Татарстан, Челябинская обл.\n2. Из 9 parent_category на Авито, наибольшее количество объявлений по категории: Личные вещи.\n3. Судя по признаку activation_date, мы имеем данные по объявлениям, которые были размещены с 15.03.2017 по 07.04.2017\n4. Наибольшее количество объявлений было размещено пользователями типа \"Private\", видимо, физические лица. Остальные две категории - \"Company\" и \"Shop\".  \n\nИзучим категории товаров подробнее.","metadata":{}},{"cell_type":"code","source":"df_train.groupby(['parent_category_name', 'category_name']).count()['user_id']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:12:19.223115Z","iopub.execute_input":"2024-11-11T11:12:19.223662Z","iopub.status.idle":"2024-11-11T11:12:21.624866Z","shell.execute_reply.started":"2024-11-11T11:12:19.223612Z","shell.execute_reply":"2024-11-11T11:12:21.623462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['param_1'].value_counts().iloc[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:12:25.77857Z","iopub.execute_input":"2024-11-11T11:12:25.78Z","iopub.status.idle":"2024-11-11T11:12:26.077754Z","shell.execute_reply.started":"2024-11-11T11:12:25.779939Z","shell.execute_reply":"2024-11-11T11:12:26.076451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = df_train['param_2'].value_counts().iloc[:10]\ndata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:12:26.172247Z","iopub.execute_input":"2024-11-11T11:12:26.172778Z","iopub.status.idle":"2024-11-11T11:12:26.391906Z","shell.execute_reply.started":"2024-11-11T11:12:26.17273Z","shell.execute_reply":"2024-11-11T11:12:26.390459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = df_train['param_3'].value_counts().iloc[:10]\ndata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:12:26.59922Z","iopub.execute_input":"2024-11-11T11:12:26.600813Z","iopub.status.idle":"2024-11-11T11:12:26.796659Z","shell.execute_reply.started":"2024-11-11T11:12:26.600752Z","shell.execute_reply":"2024-11-11T11:12:26.79521Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Можно сделать вывод, что наиболее часто встречающиеся категории - одежда и обувь, детская одежда и обувь, товары для детей, квартиры.  \n\nРассмотрим далее стоимость товаров.","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.float_format', lambda x: f'{x:.3f}')\ndf_train['price'].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:12:38.755286Z","iopub.execute_input":"2024-11-11T11:12:38.755885Z","iopub.status.idle":"2024-11-11T11:12:38.866316Z","shell.execute_reply.started":"2024-11-11T11:12:38.755821Z","shell.execute_reply":"2024-11-11T11:12:38.865007Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.histplot(np.log(df_train['price'] + 1), kde=True, bins=20, color='skyblue')\nplt.title(\"Распределение вероятности для Price\")\nplt.xlabel(\"log(Price)\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:12:40.184076Z","iopub.execute_input":"2024-11-11T11:12:40.185645Z","iopub.status.idle":"2024-11-11T11:12:48.307284Z","shell.execute_reply.started":"2024-11-11T11:12:40.185559Z","shell.execute_reply":"2024-11-11T11:12:48.305939Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Видим по перцентилям, что в целом, большинство объявлений до 7000 рублей. Можем предположить, что выше особенно дорогими объектами могут быть квартиры, машины, техника и др. Но по полю \"max\" видим, что у нас есть явно неадекватные цены. Посмотрим на них.","metadata":{}},{"cell_type":"code","source":"df_train[df_train['price'] > 100_000_000].sort_values('price', ascending = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:13:00.306138Z","iopub.execute_input":"2024-11-11T11:13:00.306733Z","iopub.status.idle":"2024-11-11T11:13:00.347148Z","shell.execute_reply.started":"2024-11-11T11:13:00.306676Z","shell.execute_reply":"2024-11-11T11:13:00.345692Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Видим, что действительно есть некоторые безумно дорогие объекты недвижимости. Ну, пусть, для остальных же объектов такая сумма больше похожа на ошибку пользователя - вместо диапазона цены, он, похоже, ввел все числа в одно поле цены. Будем иметь это ввиду, но оставим эти данные для построения модели. Ведь такие пользователи действительно есть, и нам нужно как-то их обработать. Как мы видим, вероятность совершения сделки для них очень мала, что логично, поэтому пусть модель это выучит.","metadata":{}},{"cell_type":"markdown","source":"Посмотрим, какие n-граммы из наших полей \"params\" встречаются в данных наиболее часто.","metadata":{}},{"cell_type":"code","source":"from nltk.util import ngrams\nfrom collections import Counter\n\ndf_train['params'] = df_train['param_1'].fillna('') + ' ' + df_train['param_2'].fillna('') + ' ' + df_train['param_3'].fillna('')\ndf_train['params'] = df_train['params'].str.strip()\n\ntext = ' '.join(df_train['params'].values)\ntext = [i for i in ngrams(text.lower().split(), 3)]\nprint('Common trigrams.')\nCounter(text).most_common(30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:13:50.411345Z","iopub.execute_input":"2024-11-11T11:13:50.41247Z","iopub.status.idle":"2024-11-11T11:14:02.140358Z","shell.execute_reply.started":"2024-11-11T11:13:50.412387Z","shell.execute_reply":"2024-11-11T11:14:02.138934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del text","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:14:02.143119Z","iopub.execute_input":"2024-11-11T11:14:02.144244Z","iopub.status.idle":"2024-11-11T11:14:02.44876Z","shell.execute_reply.started":"2024-11-11T11:14:02.144184Z","shell.execute_reply":"2024-11-11T11:14:02.447423Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Аналогично посмотрим для поля \"descriprion\"","metadata":{}},{"cell_type":"code","source":"from nltk.util import ngrams\nfrom collections import Counter\ndf_train['description'] = df_train['description'].apply(lambda x: str(x).replace('/\\n', ' '))\ntext = ' '.join(df_train['description'].values)\ntext = [i for i in ngrams(text.lower().split(), 3)]\nprint('Common trigrams.')\nCounter(text).most_common(40)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:14:02.450126Z","iopub.execute_input":"2024-11-11T11:14:02.451152Z","iopub.status.idle":"2024-11-11T11:15:15.897972Z","shell.execute_reply.started":"2024-11-11T11:14:02.451106Z","shell.execute_reply":"2024-11-11T11:15:15.89665Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del text","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:15:15.900972Z","iopub.execute_input":"2024-11-11T11:15:15.901479Z","iopub.status.idle":"2024-11-11T11:15:18.010854Z","shell.execute_reply.started":"2024-11-11T11:15:15.901423Z","shell.execute_reply":"2024-11-11T11:15:18.009219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\n\ng1 = sns.boxplot(x='parent_category_name',y='deal_probability', data=df_train, palette='BrBG')\ng1.set_xlabel(\"Категории\", fontsize=16)\ng1.set_ylabel('Deal Probability', fontsize=16)\ng1.set_title('Parent Category - Deal Probability', fontsize=20)\nplt.xticks(rotation=45, ha=\"right\");","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:18:38.416763Z","iopub.execute_input":"2024-11-11T11:18:38.417462Z","iopub.status.idle":"2024-11-11T11:18:41.118073Z","shell.execute_reply.started":"2024-11-11T11:18:38.417407Z","shell.execute_reply":"2024-11-11T11:18:41.116764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cm = sns.light_palette(\"#85cebb\", as_cmap=True)\npd.crosstab(df_train['parent_category_name'],\n            df_train['deal_prob_cat'],\n            normalize='index').style.background_gradient(cmap=cm)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:21:23.421896Z","iopub.execute_input":"2024-11-11T11:21:23.422421Z","iopub.status.idle":"2024-11-11T11:21:27.994992Z","shell.execute_reply.started":"2024-11-11T11:21:23.422375Z","shell.execute_reply":"2024-11-11T11:21:27.993423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['price_log'] = np.log(df_train['price'] + 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:21:39.070286Z","iopub.execute_input":"2024-11-11T11:21:39.07093Z","iopub.status.idle":"2024-11-11T11:21:39.096815Z","shell.execute_reply.started":"2024-11-11T11:21:39.070874Z","shell.execute_reply":"2024-11-11T11:21:39.09527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\n\ng1 = sns.boxplot(x='deal_prob_cat',y='price_log', data=df_train, palette='BrBG')\ng1.set_xlabel(\"Deal probability\", fontsize=16)\ng1.set_ylabel('Price Log', fontsize=16)\ng1.set_title('Price log - Deal Probability', fontsize=20)\nplt.xticks(rotation=45, ha=\"right\");","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:21:39.408132Z","iopub.execute_input":"2024-11-11T11:21:39.40882Z","iopub.status.idle":"2024-11-11T11:21:40.506643Z","shell.execute_reply.started":"2024-11-11T11:21:39.408759Z","shell.execute_reply":"2024-11-11T11:21:40.505399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'Видим, что наибольшая вероятность сделки наблюдается тогда, когда цена находится в диапазоне {round(np.exp(5),2)}-{round(np.exp(10),2)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:21:47.054956Z","iopub.execute_input":"2024-11-11T11:21:47.055818Z","iopub.status.idle":"2024-11-11T11:21:47.069542Z","shell.execute_reply.started":"2024-11-11T11:21:47.055724Z","shell.execute_reply":"2024-11-11T11:21:47.066544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\n\ng1 = sns.boxplot(x='region',y='deal_probability', data=df_train, palette='BrBG')\ng1.set_xlabel(\"Категории\", fontsize=16)\ng1.set_ylabel('Deal Probability', fontsize=16)\ng1.set_title('Region - Deal Probability', fontsize=20)\nplt.xticks(rotation=45, ha=\"right\");","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T11:21:48.225859Z","iopub.execute_input":"2024-11-11T11:21:48.226418Z","iopub.status.idle":"2024-11-11T11:21:51.611601Z","shell.execute_reply.started":"2024-11-11T11:21:48.22637Z","shell.execute_reply":"2024-11-11T11:21:51.610302Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Перейдем к созданию модели","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/avito-demand-prediction/train.csv\")\ndf_test = pd.read_csv(\"../input/avito-demand-prediction/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:54:55.240133Z","iopub.execute_input":"2024-11-11T12:54:55.240519Z","iopub.status.idle":"2024-11-11T12:55:20.807371Z","shell.execute_reply.started":"2024-11-11T12:54:55.240484Z","shell.execute_reply":"2024-11-11T12:55:20.806532Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Переведем наш признак \"title\" в представление TF-IDF","metadata":{}},{"cell_type":"code","source":"from nltk.corpus import stopwords\n\nstopWords = list(stopwords.words('russian'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:20.80964Z","iopub.execute_input":"2024-11-11T12:55:20.810252Z","iopub.status.idle":"2024-11-11T12:55:20.815271Z","shell.execute_reply.started":"2024-11-11T12:55:20.8102Z","shell.execute_reply":"2024-11-11T12:55:20.814345Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nvectorizer = TfidfVectorizer(stop_words=stopWords, max_features=2000)\nvectorizer.fit(df_train['title'])\n\ntrain_title = vectorizer.transform(df_train['title'])\ntest_title = vectorizer.transform(df_test['title'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:20.816403Z","iopub.execute_input":"2024-11-11T12:55:20.816755Z","iopub.status.idle":"2024-11-11T12:55:52.365603Z","shell.execute_reply.started":"2024-11-11T12:55:20.816706Z","shell.execute_reply":"2024-11-11T12:55:52.364566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_title_df = pd.DataFrame.sparse.from_spmatrix(train_title, columns=vectorizer.get_feature_names_out())\ntest_title_df = pd.DataFrame.sparse.from_spmatrix(test_title, columns=vectorizer.get_feature_names_out())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:52.367913Z","iopub.execute_input":"2024-11-11T12:55:52.368261Z","iopub.status.idle":"2024-11-11T12:55:52.481881Z","shell.execute_reply.started":"2024-11-11T12:55:52.368226Z","shell.execute_reply":"2024-11-11T12:55:52.48114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rmse(predictions, targets):\n    return np.sqrt(((predictions - targets) ** 2).mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:52.483023Z","iopub.execute_input":"2024-11-11T12:55:52.483341Z","iopub.status.idle":"2024-11-11T12:55:52.488322Z","shell.execute_reply.started":"2024-11-11T12:55:52.483308Z","shell.execute_reply":"2024-11-11T12:55:52.487247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Перейдем к обработке пропущенных значений.","metadata":{}},{"cell_type":"code","source":"df_train['price'] = df_train['price'].fillna(df_train['price'].mean())\ndf_test['price'] = df_test['price'].fillna(df_train['price'].mean())\n\ndf_test.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:52.489448Z","iopub.execute_input":"2024-11-11T12:55:52.48974Z","iopub.status.idle":"2024-11-11T12:55:53.177964Z","shell.execute_reply.started":"2024-11-11T12:55:52.489704Z","shell.execute_reply":"2024-11-11T12:55:53.176962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in ['param_1', 'param_2', 'param_3', 'image_top_1', 'title', 'description']:\n    df_train[col] = df_train[col].fillna('')\n    df_test[col] = df_test[col].fillna('')\n    \ndf_test.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:53.179096Z","iopub.execute_input":"2024-11-11T12:55:53.179411Z","iopub.status.idle":"2024-11-11T12:55:55.329253Z","shell.execute_reply.started":"2024-11-11T12:55:53.17938Z","shell.execute_reply":"2024-11-11T12:55:55.328257Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# В авито это некоторый рассчитанный показатель \"хорошей\" картинки, будем считать его как категориальную переменную\ndf_train['image_top_1'] = df_train['image_top_1'].astype('str')\ndf_test['image_top_1'] = df_test['image_top_1'].astype('str')    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:55.330419Z","iopub.execute_input":"2024-11-11T12:55:55.330733Z","iopub.status.idle":"2024-11-11T12:55:56.180343Z","shell.execute_reply.started":"2024-11-11T12:55:55.330701Z","shell.execute_reply":"2024-11-11T12:55:56.179431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_features = ['region', 'city', 'parent_category_name', 'category_name', 'param_1', 'param_2', 'param_3', 'user_type', 'image_top_1']\ntext_features = ['title', 'description']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:56.181792Z","iopub.execute_input":"2024-11-11T12:55:56.18252Z","iopub.status.idle":"2024-11-11T12:55:56.187296Z","shell.execute_reply.started":"2024-11-11T12:55:56.182473Z","shell.execute_reply":"2024-11-11T12:55:56.186338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Удаляем не особо значимые признаки\n\ndf_train.drop(['image', 'item_id', 'user_id', 'activation_date'], axis=1, inplace=True)\ndf_test.drop(['image', 'item_id', 'user_id', 'activation_date'], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:56.190247Z","iopub.execute_input":"2024-11-11T12:55:56.190593Z","iopub.status.idle":"2024-11-11T12:55:56.861735Z","shell.execute_reply.started":"2024-11-11T12:55:56.190555Z","shell.execute_reply":"2024-11-11T12:55:56.860935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:35:17.209066Z","iopub.execute_input":"2024-11-11T12:35:17.209398Z","iopub.status.idle":"2024-11-11T12:35:17.230112Z","shell.execute_reply.started":"2024-11-11T12:35:17.209364Z","shell.execute_reply":"2024-11-11T12:35:17.229218Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Проведем пока обучение без TF-IDF**","metadata":{}},{"cell_type":"code","source":"from scipy.sparse import hstack, csr_matrix\nfrom sklearn.model_selection import train_test_split\n\n\nX = df_train.drop(columns=['deal_probability', 'title', 'description'])\nX_test = df_test.drop(columns=['title', 'description'])\n\ny = df_train['deal_probability']\nX.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:35:20.646605Z","iopub.execute_input":"2024-11-11T12:35:20.647443Z","iopub.status.idle":"2024-11-11T12:35:20.874022Z","shell.execute_reply.started":"2024-11-11T12:35:20.64739Z","shell.execute_reply":"2024-11-11T12:35:20.872982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Данные очень большие, один из варинтов того, как обучаться не на всех данных сразу - обучаться фолдами\n# В каком-то смысле оборачиваем бустинг в бэггинг :)\nfrom sklearn.model_selection import StratifiedKFold\n\n\nspliter = StratifiedKFold(n_splits=5, shuffle=True,\n                          random_state=3)\n\n_y = (df_train.deal_probability.round(2)*100).astype(int) # Делаеам так, чтобы отработал K-Fold, как бы разбиваем на 100 классов\n\nFOLD_LIST = list(spliter.split(_y, _y))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:35:22.687452Z","iopub.execute_input":"2024-11-11T12:35:22.688309Z","iopub.status.idle":"2024-11-11T12:35:23.116939Z","shell.execute_reply.started":"2024-11-11T12:35:22.688259Z","shell.execute_reply":"2024-11-11T12:35:23.115919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from catboost import CatBoostRegressor, Pool\nfrom tqdm import tqdm_notebook\n# Вообще, помимо категориальных переменных, CB может работать и с текстовыми данными\n# Но для задачи регрессии эта фича не реализована, пока только для классификации\n\n_models = []\n\noof_predictions = np.zeros(shape=[X.shape[0]])\n\nfor fold_id, (train_idx, val_idx) in tqdm_notebook(enumerate(FOLD_LIST)):\n    \n    X_train, Y_train = X.loc[train_idx], y.loc[train_idx]\n    X_val, Y_val = X.loc[val_idx], y.loc[val_idx]\n    \n    train_dataset = Pool(X_train, Y_train,\n                         cat_features=cat_features)\n    \n    eval_dataset = Pool(X_val, Y_val,\n                        cat_features=cat_features)\n    \n    model = CatBoostRegressor(\n        learning_rate=0.1, iterations=1000, eval_metric='RMSE',\n        metric_period=50, early_stopping_rounds=20, task_type=\"GPU\",\n    )\n    model.fit(train_dataset, eval_set=eval_dataset)\n    \n    _models.append(model)\n    preds = model.predict(X_val)\n    oof_predictions[val_idx] += preds\n    print('fold_id:', fold_id, rmse(Y_val, preds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:35:42.059406Z","iopub.execute_input":"2024-11-11T12:35:42.059932Z","iopub.status.idle":"2024-11-11T12:47:31.533551Z","shell.execute_reply.started":"2024-11-11T12:35:42.059894Z","shell.execute_reply":"2024-11-11T12:47:31.532537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof_predictions /= len(FOLD_LIST)\noof_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:49:19.611937Z","iopub.execute_input":"2024-11-11T12:49:19.612807Z","iopub.status.idle":"2024-11-11T12:49:19.620216Z","shell.execute_reply.started":"2024-11-11T12:49:19.612766Z","shell.execute_reply":"2024-11-11T12:49:19.619134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rmse(oof_predictions, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:49:26.078474Z","iopub.execute_input":"2024-11-11T12:49:26.079177Z","iopub.status.idle":"2024-11-11T12:49:26.107646Z","shell.execute_reply.started":"2024-11-11T12:49:26.079137Z","shell.execute_reply":"2024-11-11T12:49:26.10678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Теперь проведем с использовнием TF-IDF**","metadata":{}},{"cell_type":"code","source":"df_train.rename(columns={'price': 'ad_price'}, inplace=True)\ndf_test.rename(columns={'price': 'ad_price'}, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:56.86284Z","iopub.execute_input":"2024-11-11T12:55:56.863184Z","iopub.status.idle":"2024-11-11T12:55:56.868539Z","shell.execute_reply.started":"2024-11-11T12:55:56.863151Z","shell.execute_reply":"2024-11-11T12:55:56.867678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train_tfidf_title = pd.concat([df_train, train_title_df], axis=1)\ndf_train_tfidf_title.shape\ndel train_title_df\n\ndf_test_tfidf_title = pd.concat([df_test, test_title_df], axis=1)\ndf_test_tfidf_title.shape\ndel test_title_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:56.869696Z","iopub.execute_input":"2024-11-11T12:55:56.870027Z","iopub.status.idle":"2024-11-11T12:55:58.284706Z","shell.execute_reply.started":"2024-11-11T12:55:56.869991Z","shell.execute_reply":"2024-11-11T12:55:58.283877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df_train_tfidf_title.drop(columns=['deal_probability', 'title', 'description'])\nX_test = df_test_tfidf_title.drop(columns=['title', 'description'])\n\ny = df_train_tfidf_title['deal_probability']\ndel df_train_tfidf_title, df_test_tfidf_title","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:58.285868Z","iopub.execute_input":"2024-11-11T12:55:58.286189Z","iopub.status.idle":"2024-11-11T12:55:58.726666Z","shell.execute_reply.started":"2024-11-11T12:55:58.286156Z","shell.execute_reply":"2024-11-11T12:55:58.725865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from catboost import CatBoostRegressor, Pool\nfrom tqdm import tqdm_notebook\n\n\n_models_tfidf = []\n\noof_predictions = np.zeros(shape=[X.shape[0]])\n\nfor fold_id, (train_idx, val_idx) in tqdm_notebook(enumerate(FOLD_LIST)):\n    \n    X_train, Y_train = X.loc[train_idx], y.loc[train_idx]\n    X_val, Y_val = X.loc[val_idx], y.loc[val_idx]\n    \n    train_dataset = Pool(X_train, Y_train,\n                         cat_features=cat_features)\n    \n    eval_dataset = Pool(X_val, Y_val,\n                        cat_features=cat_features)\n    \n    model = CatBoostRegressor(\n        learning_rate=0.1, iterations=1000, eval_metric='RMSE',\n        metric_period=50, early_stopping_rounds=20, task_type=\"GPU\",\n    )\n    model.fit(train_dataset, eval_set=eval_dataset)\n    \n    _models_tfidf.append(model)\n    preds = model.predict(X_val)\n    oof_predictions[val_idx] += preds\n    print('fold_id:', fold_id, rmse(Y_val, preds))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T12:55:58.729254Z","iopub.execute_input":"2024-11-11T12:55:58.729548Z","iopub.status.idle":"2024-11-11T13:18:28.144289Z","shell.execute_reply.started":"2024-11-11T12:55:58.729518Z","shell.execute_reply":"2024-11-11T13:18:28.143249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof_predictions /= len(FOLD_LIST)\noof_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T13:23:23.537559Z","iopub.execute_input":"2024-11-11T13:23:23.538012Z","iopub.status.idle":"2024-11-11T13:23:23.545919Z","shell.execute_reply.started":"2024-11-11T13:23:23.537945Z","shell.execute_reply":"2024-11-11T13:23:23.54489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rmse(oof_predictions, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T13:23:23.819094Z","iopub.execute_input":"2024-11-11T13:23:23.819505Z","iopub.status.idle":"2024-11-11T13:23:23.849611Z","shell.execute_reply.started":"2024-11-11T13:23:23.819466Z","shell.execute_reply":"2024-11-11T13:23:23.84876Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Подготовим Submission**","metadata":{}},{"cell_type":"code","source":"pred = np.zeros(shape=[X_test.shape[0]])\n\nfor model in tqdm_notebook(_models):\n# for model in tqdm_notebook(_models_tfidf):\n    preds = model.predict(X_test)\n    pred += preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T13:23:59.961066Z","iopub.execute_input":"2024-11-11T13:23:59.961492Z","iopub.status.idle":"2024-11-11T13:24:25.324116Z","shell.execute_reply.started":"2024-11-11T13:23:59.961449Z","shell.execute_reply":"2024-11-11T13:24:25.323131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred /= len(_models_tfidf)\npred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T13:25:10.539018Z","iopub.execute_input":"2024-11-11T13:25:10.539404Z","iopub.status.idle":"2024-11-11T13:25:10.546495Z","shell.execute_reply.started":"2024-11-11T13:25:10.539369Z","shell.execute_reply":"2024-11-11T13:25:10.545356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv('../input/avito-demand-prediction/sample_submission.csv')\nsample_sub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T13:25:36.567875Z","iopub.execute_input":"2024-11-11T13:25:36.56828Z","iopub.status.idle":"2024-11-11T13:25:36.923682Z","shell.execute_reply.started":"2024-11-11T13:25:36.568243Z","shell.execute_reply":"2024-11-11T13:25:36.922741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub['deal_probability'] = np.clip(pred, 0, 1)\nsample_sub.to_csv('sub.csv', index=False)\nsample_sub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T13:25:41.769605Z","iopub.execute_input":"2024-11-11T13:25:41.770012Z","iopub.status.idle":"2024-11-11T13:25:43.233973Z","shell.execute_reply.started":"2024-11-11T13:25:41.769965Z","shell.execute_reply":"2024-11-11T13:25:43.232893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-11T13:26:10.3979Z","iopub.execute_input":"2024-11-11T13:26:10.39864Z","iopub.status.idle":"2024-11-11T13:26:11.965135Z","shell.execute_reply.started":"2024-11-11T13:26:10.3986Z","shell.execute_reply":"2024-11-11T13:26:11.964298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}