{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom statsmodels.graphics.gofplots import qqplot\nimport xgboost as xgb\nfrom sklearn import preprocessing\nimport os\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-04T14:08:23.552842Z","iopub.execute_input":"2022-08-04T14:08:23.553165Z","iopub.status.idle":"2022-08-04T14:08:25.111628Z","shell.execute_reply.started":"2022-08-04T14:08:23.553081Z","shell.execute_reply":"2022-08-04T14:08:25.110522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_oil = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/oil.csv')\ndat_oil","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:25.113167Z","iopub.execute_input":"2022-08-04T14:08:25.113499Z","iopub.status.idle":"2022-08-04T14:08:25.144962Z","shell.execute_reply.started":"2022-08-04T14:08:25.113465Z","shell.execute_reply":"2022-08-04T14:08:25.143792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_oil.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:25.146777Z","iopub.execute_input":"2022-08-04T14:08:25.147091Z","iopub.status.idle":"2022-08-04T14:08:25.173516Z","shell.execute_reply.started":"2022-08-04T14:08:25.147049Z","shell.execute_reply":"2022-08-04T14:08:25.172324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_simple_submission = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/sample_submission.csv')\ndat_simple_submission","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:25.175727Z","iopub.execute_input":"2022-08-04T14:08:25.176053Z","iopub.status.idle":"2022-08-04T14:08:25.205281Z","shell.execute_reply.started":"2022-08-04T14:08:25.176010Z","shell.execute_reply":"2022-08-04T14:08:25.204280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_simple_submission.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:25.206641Z","iopub.execute_input":"2022-08-04T14:08:25.207232Z","iopub.status.idle":"2022-08-04T14:08:25.221435Z","shell.execute_reply.started":"2022-08-04T14:08:25.207188Z","shell.execute_reply":"2022-08-04T14:08:25.220450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_holidays_events = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/holidays_events.csv')\ndat_holidays_events","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:25.222922Z","iopub.execute_input":"2022-08-04T14:08:25.223420Z","iopub.status.idle":"2022-08-04T14:08:25.245747Z","shell.execute_reply.started":"2022-08-04T14:08:25.223377Z","shell.execute_reply":"2022-08-04T14:08:25.244920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_holidays_events.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:25.246883Z","iopub.execute_input":"2022-08-04T14:08:25.247452Z","iopub.status.idle":"2022-08-04T14:08:25.260173Z","shell.execute_reply.started":"2022-08-04T14:08:25.247417Z","shell.execute_reply":"2022-08-04T14:08:25.259206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_stores = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/stores.csv')\ndat_stores","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:25.261401Z","iopub.execute_input":"2022-08-04T14:08:25.261654Z","iopub.status.idle":"2022-08-04T14:08:25.289033Z","shell.execute_reply.started":"2022-08-04T14:08:25.261624Z","shell.execute_reply":"2022-08-04T14:08:25.288398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_stores.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:25.289968Z","iopub.execute_input":"2022-08-04T14:08:25.290522Z","iopub.status.idle":"2022-08-04T14:08:25.302510Z","shell.execute_reply.started":"2022-08-04T14:08:25.290476Z","shell.execute_reply":"2022-08-04T14:08:25.301776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_train = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/train.csv')\ndat_train","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:25.305180Z","iopub.execute_input":"2022-08-04T14:08:25.305793Z","iopub.status.idle":"2022-08-04T14:08:28.072715Z","shell.execute_reply.started":"2022-08-04T14:08:25.305725Z","shell.execute_reply":"2022-08-04T14:08:28.071837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:28.074164Z","iopub.execute_input":"2022-08-04T14:08:28.074607Z","iopub.status.idle":"2022-08-04T14:08:28.086292Z","shell.execute_reply.started":"2022-08-04T14:08:28.074548Z","shell.execute_reply":"2022-08-04T14:08:28.085416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_test = pd.read_csv('/kaggle/input/store-sales-time-series-forecasting/test.csv')\ndat_test","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:28.087967Z","iopub.execute_input":"2022-08-04T14:08:28.088729Z","iopub.status.idle":"2022-08-04T14:08:28.137434Z","shell.execute_reply.started":"2022-08-04T14:08:28.088678Z","shell.execute_reply":"2022-08-04T14:08:28.136557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:28.138765Z","iopub.execute_input":"2022-08-04T14:08:28.139018Z","iopub.status.idle":"2022-08-04T14:08:28.155088Z","shell.execute_reply.started":"2022-08-04T14:08:28.138983Z","shell.execute_reply":"2022-08-04T14:08:28.154209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sns.PairGrid(dat_train).map(sns.scatterplot) #попарный график зависимости","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:28.156020Z","iopub.execute_input":"2022-08-04T14:08:28.156260Z","iopub.status.idle":"2022-08-04T14:08:28.160522Z","shell.execute_reply.started":"2022-08-04T14:08:28.156229Z","shell.execute_reply":"2022-08-04T14:08:28.159763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datv1 = dat_train.merge(dat_oil)\ndatv1","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:28.161376Z","iopub.execute_input":"2022-08-04T14:08:28.161622Z","iopub.status.idle":"2022-08-04T14:08:28.791199Z","shell.execute_reply.started":"2022-08-04T14:08:28.161591Z","shell.execute_reply":"2022-08-04T14:08:28.790406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datv2 = datv1.merge(dat_stores)\ndatv2.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:28.792436Z","iopub.execute_input":"2022-08-04T14:08:28.792642Z","iopub.status.idle":"2022-08-04T14:08:29.284212Z","shell.execute_reply.started":"2022-08-04T14:08:28.792616Z","shell.execute_reply":"2022-08-04T14:08:29.283238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dat_v2 = pd.merge_orderer(dat, dat_stores, fill_method=)\n#dat_v2","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:29.285434Z","iopub.execute_input":"2022-08-04T14:08:29.285653Z","iopub.status.idle":"2022-08-04T14:08:29.290388Z","shell.execute_reply.started":"2022-08-04T14:08:29.285626Z","shell.execute_reply":"2022-08-04T14:08:29.289413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datv2.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:29.291737Z","iopub.execute_input":"2022-08-04T14:08:29.292050Z","iopub.status.idle":"2022-08-04T14:08:30.143134Z","shell.execute_reply.started":"2022-08-04T14:08:29.292016Z","shell.execute_reply":"2022-08-04T14:08:30.142281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(datv2.corr(), cmap=\"YlGnBu\") #График корелляции","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:30.144157Z","iopub.execute_input":"2022-08-04T14:08:30.144392Z","iopub.status.idle":"2022-08-04T14:08:30.805287Z","shell.execute_reply.started":"2022-08-04T14:08:30.144350Z","shell.execute_reply":"2022-08-04T14:08:30.804418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datv2.drop(['date', 'store_nbr', 'state'], axis=1, inplace=True)\ndatv2","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:30.806757Z","iopub.execute_input":"2022-08-04T14:08:30.807079Z","iopub.status.idle":"2022-08-04T14:08:30.977291Z","shell.execute_reply.started":"2022-08-04T14:08:30.807036Z","shell.execute_reply":"2022-08-04T14:08:30.976606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datv2['family'] #смотрим нужный нам столбик","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:30.978479Z","iopub.execute_input":"2022-08-04T14:08:30.978843Z","iopub.status.idle":"2022-08-04T14:08:30.987262Z","shell.execute_reply.started":"2022-08-04T14:08:30.978812Z","shell.execute_reply":"2022-08-04T14:08:30.986164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lenc_fam = preprocessing.LabelEncoder() #создаем функцию превращаю категориальные данные в числовые\nlenc_fam.fit(datv2['family']) #обучаем ее на нашем столбце ","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:30.988790Z","iopub.execute_input":"2022-08-04T14:08:30.989131Z","iopub.status.idle":"2022-08-04T14:08:31.082032Z","shell.execute_reply.started":"2022-08-04T14:08:30.989087Z","shell.execute_reply":"2022-08-04T14:08:31.081004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lenc_fam.classes_","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:31.083501Z","iopub.execute_input":"2022-08-04T14:08:31.083786Z","iopub.status.idle":"2022-08-04T14:08:31.091263Z","shell.execute_reply.started":"2022-08-04T14:08:31.083728Z","shell.execute_reply":"2022-08-04T14:08:31.090521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datv2['family'] = lenc_fam.transform(datv2['family']) #заменяем значения в массиве на числовые для столбца family","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:31.092585Z","iopub.execute_input":"2022-08-04T14:08:31.093404Z","iopub.status.idle":"2022-08-04T14:08:31.624814Z","shell.execute_reply.started":"2022-08-04T14:08:31.093326Z","shell.execute_reply":"2022-08-04T14:08:31.623944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lenc_city = preprocessing.LabelEncoder() #создаем функцию превращаю категориальные данные в числовые\nlenc_city.fit(datv2['city']) #обучаем ее на нашем столбце \ndatv2['city'] = lenc_city.transform(datv2['city']) #заменяем значения в массиве на числовые для столбца city","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:31.626060Z","iopub.execute_input":"2022-08-04T14:08:31.626372Z","iopub.status.idle":"2022-08-04T14:08:32.153881Z","shell.execute_reply.started":"2022-08-04T14:08:31.626326Z","shell.execute_reply":"2022-08-04T14:08:32.153129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lenc_type = preprocessing.LabelEncoder() #создаем функцию превращаю категориальные данные в числовые\nlenc_type.fit(datv2['type']) #обучаем ее на нашем столбце \ndatv2['type'] = lenc_type.transform(datv2['type']) #заменяем значения в массиве на числовые для столбца type","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:32.155122Z","iopub.execute_input":"2022-08-04T14:08:32.155478Z","iopub.status.idle":"2022-08-04T14:08:32.649978Z","shell.execute_reply.started":"2022-08-04T14:08:32.155434Z","shell.execute_reply":"2022-08-04T14:08:32.649197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Ищем выбросы с помощью плотбоксов","metadata":{}},{"cell_type":"code","source":"colnams = datv2.columns\nfor colname in colnams:\n    print(colname)\n    sub_data = datv2[colname]\n    \n    #Диаграмма \"Ящик с усами\"\n    sns.boxplot(x=sub_data);\n    plt.show()\n    \n    #Значения (мин макс ср)\n    print(\"Минимальное значение: \", end=\" \")\n    print(np.min(sub_data))\n    print(\"Максимальное значение: \", end=\" \")\n    print(np.max(sub_data))\n    print(\"Среднее значение: \", end=\" \")\n    print(np.mean(sub_data))\n\n    print(\"Медианное значение: \", end=\" \")\n    print(np.median(sub_data))\n    print(\"\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:32.651709Z","iopub.execute_input":"2022-08-04T14:08:32.652349Z","iopub.status.idle":"2022-08-04T14:08:36.280890Z","shell.execute_reply.started":"2022-08-04T14:08:32.652298Z","shell.execute_reply":"2022-08-04T14:08:36.279935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Удалим выбросы в вкладке sales","metadata":{}},{"cell_type":"code","source":"datv2.loc[datv2['sales'] > 13000, 'sales'] = np.nan\ndatv2.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:36.284671Z","iopub.execute_input":"2022-08-04T14:08:36.284933Z","iopub.status.idle":"2022-08-04T14:08:36.336126Z","shell.execute_reply.started":"2022-08-04T14:08:36.284897Z","shell.execute_reply":"2022-08-04T14:08:36.335449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datv3 = datv2.dropna(axis=0) #удаляем выбросы\ndatv3.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:36.337582Z","iopub.execute_input":"2022-08-04T14:08:36.338164Z","iopub.status.idle":"2022-08-04T14:08:36.597955Z","shell.execute_reply.started":"2022-08-04T14:08:36.338129Z","shell.execute_reply":"2022-08-04T14:08:36.597107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colnams2 = datv2.columns\nfor colname2 in colnams2:\n    print(colname2)\n    sub_data2 = datv2[colname2]\n    \n    #Диаграмма \"Ящик с усами\"\n    sns.boxplot(x=sub_data2);\n    plt.show()\n    \n    #Значения (мин макс ср)\n    print(\"Минимальное значение: \", end=\" \")\n    print(np.min(sub_data2))\n    print(\"Максимальное значение: \", end=\" \")\n    print(np.max(sub_data2))\n    print(\"Среднее значение: \", end=\" \")\n    print(np.mean(sub_data2))\n\n    print(\"Медианное значение: \", end=\" \")\n    print(np.median(sub_data2))\n    print(\"\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:36.599277Z","iopub.execute_input":"2022-08-04T14:08:36.599622Z","iopub.status.idle":"2022-08-04T14:08:40.102103Z","shell.execute_reply.started":"2022-08-04T14:08:36.599579Z","shell.execute_reply":"2022-08-04T14:08:40.101177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Нормализация данных MinMaxScaler'ом","metadata":{}},{"cell_type":"code","source":"minmaxscalar = preprocessing.MinMaxScaler()\ncol = datv3.columns\nresult = minmaxscalar.fit_transform(datv3)\ndata_norm = pd.DataFrame(result, columns=col)\ndata_norm","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:40.103421Z","iopub.execute_input":"2022-08-04T14:08:40.103698Z","iopub.status.idle":"2022-08-04T14:08:40.373732Z","shell.execute_reply.started":"2022-08-04T14:08:40.103665Z","shell.execute_reply":"2022-08-04T14:08:40.372923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_norm.isnull().sum() #смотрим пустые значения","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:40.374766Z","iopub.execute_input":"2022-08-04T14:08:40.374981Z","iopub.status.idle":"2022-08-04T14:08:40.415194Z","shell.execute_reply.started":"2022-08-04T14:08:40.374954Z","shell.execute_reply":"2022-08-04T14:08:40.414413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_norm.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:40.416308Z","iopub.execute_input":"2022-08-04T14:08:40.416584Z","iopub.status.idle":"2022-08-04T14:08:40.426979Z","shell.execute_reply.started":"2022-08-04T14:08:40.416554Z","shell.execute_reply":"2022-08-04T14:08:40.426320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_pure = data_norm.dropna(axis=0) #удаляем выбросы\ndata_pure.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:40.427909Z","iopub.execute_input":"2022-08-04T14:08:40.428843Z","iopub.status.idle":"2022-08-04T14:08:40.645212Z","shell.execute_reply.started":"2022-08-04T14:08:40.428804Z","shell.execute_reply":"2022-08-04T14:08:40.644295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Построение моделей","metadata":{}},{"cell_type":"markdown","source":"Загрузка библиотек","metadata":{}},{"cell_type":"code","source":"from keras.wrappers.scikit_learn import KerasClassifier\nfrom keras.models import Sequential\nfrom keras.layers import Activation, Dropout, Dense, Flatten\nfrom keras.losses import SparseCategoricalCrossentropy\n\nfrom numpy.random import seed\n\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.linear_model import LinearRegression, LogisticRegression\nfrom sklearn.svm import SVR\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom sklearn.metrics import r2_score\nfrom sklearn.model_selection import train_test_split, GridSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:40.647093Z","iopub.execute_input":"2022-08-04T14:08:40.647616Z","iopub.status.idle":"2022-08-04T14:08:46.796422Z","shell.execute_reply.started":"2022-08-04T14:08:40.647570Z","shell.execute_reply":"2022-08-04T14:08:46.795432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_colnames = ['family',\n                 'onpromotion',\n                 'dcoilwtico',\n                 'city',\n                 'type',\n                 'cluster']\noutput_colnames = ['sales']","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:46.797758Z","iopub.execute_input":"2022-08-04T14:08:46.798039Z","iopub.status.idle":"2022-08-04T14:08:46.803381Z","shell.execute_reply.started":"2022-08-04T14:08:46.798005Z","shell.execute_reply":"2022-08-04T14:08:46.802511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"in_train_mod = data_pure[input_colnames] #Массив на вход\nout_train_mod = data_pure[output_colnames] #Массив на выход\nout_train_mod","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:46.805107Z","iopub.execute_input":"2022-08-04T14:08:46.805801Z","iopub.status.idle":"2022-08-04T14:08:46.874471Z","shell.execute_reply.started":"2022-08-04T14:08:46.805750Z","shell.execute_reply":"2022-08-04T14:08:46.873521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Подготовка тестовой и обучающей выборки","metadata":{}},{"cell_type":"code","source":"train_in, test_in, train_out, test_out = train_test_split(in_train_mod, out_train_mod, test_size=0.3)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:46.875846Z","iopub.execute_input":"2022-08-04T14:08:46.876056Z","iopub.status.idle":"2022-08-04T14:08:47.327041Z","shell.execute_reply.started":"2022-08-04T14:08:46.876029Z","shell.execute_reply":"2022-08-04T14:08:47.326122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Построение нескольких моделей для прогноза параметра sales.","metadata":{}},{"cell_type":"code","source":"models = [RandomForestRegressor(n_estimators=87), # случайный лес\n          KNeighborsRegressor(n_neighbors=21)] # метод ближайших соседей","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:47.328418Z","iopub.execute_input":"2022-08-04T14:08:47.328678Z","iopub.status.idle":"2022-08-04T14:08:47.333124Z","shell.execute_reply.started":"2022-08-04T14:08:47.328647Z","shell.execute_reply":"2022-08-04T14:08:47.332017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models_prov = pd.DataFrame() #создаем двумерный массив\nmodels_results = pd.DataFrame()\ntmp = {} #временный словарь\ntmp2 = {}\nfor model in models:      #цикл перебора моделей\n    m = str(model)\n    tmp['Model'] = m[:m.index('(')]\n    tmp2['Model'] = m[:m.index('(')]\n    \n    for i in range(train_out.shape[1]): #цикл проверки модели\n        model.fit(train_in, train_out[output_colnames[i]])  #обучаем модель\n        tmp['R2_Y%s'%str(i+1)] = r2_score(test_out[output_colnames[i]], model.predict(test_in))#считаем коэффициент детерминации\n        #Создадим датасет с тестовыми данными и прогнозными значениями для каждой модели\n        tmp2 = pd.DataFrame({'Данные': test_out[output_colnames[i]],'Прогноз': model.predict(test_in)})   \n        \n    models_prov = models_prov.append([tmp]) #добавляем данные и итоговый массив\n    models_results = tmp2 \n    #Выведем график рассеяния. В случае идеального прогноза график походил бы на прямую\n    sns.set_style('darkgrid')\n    plt.title( model, size=16)\n    plt.xlabel('Данные',size=12)\n    plt.ylabel('Прогноз',size=12)\n    plt.ylim(0, 1)\n    plt.xlim(0, 1)\n    sns.scatterplot(x='Данные', y='Прогноз', data=models_results, edgecolor='black', palette='cubehelix')\n    plt.show()\n    \nmodels_prov.set_index('Model', inplace=True) #делаем индекс по названию модели\nmodels_prov","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:08:47.334275Z","iopub.execute_input":"2022-08-04T14:08:47.334519Z","iopub.status.idle":"2022-08-04T14:18:44.630705Z","shell.execute_reply.started":"2022-08-04T14:08:47.334490Z","shell.execute_reply":"2022-08-04T14:18:44.629696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#invert_scalar = minmaxscalar.inverse_transform(X)\n#data_norm = pd.DataFrame(result, columns=col)\n#data_norm","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:18:44.631926Z","iopub.execute_input":"2022-08-04T14:18:44.632164Z","iopub.status.idle":"2022-08-04T14:18:44.636794Z","shell.execute_reply.started":"2022-08-04T14:18:44.632134Z","shell.execute_reply":"2022-08-04T14:18:44.635904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_test","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:18:44.637879Z","iopub.execute_input":"2022-08-04T14:18:44.638115Z","iopub.status.idle":"2022-08-04T14:18:44.663272Z","shell.execute_reply.started":"2022-08-04T14:18:44.638086Z","shell.execute_reply":"2022-08-04T14:18:44.662169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:18:44.664764Z","iopub.execute_input":"2022-08-04T14:18:44.665258Z","iopub.status.idle":"2022-08-04T14:18:44.686718Z","shell.execute_reply.started":"2022-08-04T14:18:44.665214Z","shell.execute_reply":"2022-08-04T14:18:44.685780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_test['family'] = lenc_fam.transform(dat_test['family']) #заменяем значения в массиве на числовые для столбца family\ndat_test","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:18:44.688109Z","iopub.execute_input":"2022-08-04T14:18:44.689012Z","iopub.status.idle":"2022-08-04T14:18:44.713561Z","shell.execute_reply.started":"2022-08-04T14:18:44.688964Z","shell.execute_reply":"2022-08-04T14:18:44.712586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dat_test.drop(['date'], axis=1, inplace=True)\ndat_test","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:18:44.715107Z","iopub.execute_input":"2022-08-04T14:18:44.715951Z","iopub.status.idle":"2022-08-04T14:18:44.731135Z","shell.execute_reply.started":"2022-08-04T14:18:44.715907Z","shell.execute_reply":"2022-08-04T14:18:44.730307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#new_predict = models[0].predict(dat_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T14:18:44.732246Z","iopub.execute_input":"2022-08-04T14:18:44.732925Z","iopub.status.idle":"2022-08-04T14:18:44.737138Z","shell.execute_reply.started":"2022-08-04T14:18:44.732870Z","shell.execute_reply":"2022-08-04T14:18:44.736375Z"},"trusted":true},"execution_count":null,"outputs":[]}]}