{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/avito-demand-prediction/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder, OneHotEncoder\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.linear_model import Ridge\n\nimport re\n\nfrom scipy.sparse import hstack, csr_matrix\n\nfrom tqdm import tqdm_notebook\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"spliter = StratifiedKFold(n_splits=5, shuffle=True)\n\n_y = (train.deal_probability.round(2)*100).astype(int)\n\nFOLD_LIST = list(spliter.split(_y, _y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def text_preprocession(text):\n    text = str(text)\n    text = text.lower()\n    clean = re.sub(r\"[,.;@#?!&$]+\\ *\", \" \", text)\n    return clean","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"Y = train.deal_probability.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.description = train.description.apply(text_preprocession)\ntrain.title = train.title.apply(text_preprocession)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"description_vectorizer = TfidfVectorizer(max_df=0.9, min_df=7, max_features=50000)\n\ntitle_vectorizer = TfidfVectorizer(max_df=0.9, analyzer='char', ngram_range=(3,3), min_df=7, max_features=50000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_description_tfidf = description_vectorizer.fit_transform(train.description )\ntrain_title_tfidf = title_vectorizer.fit_transform(train.title)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.DataFrame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"categorial_features = [\n    'user_type',\n    'image_top_1',\n    'region',\n    'city',\n    'parent_category_name',\n    'category_name',\n    'param_1',\n    'param_2',\n    'param_3'\n    \n]\nlabel_encoder_list = []\n\n\nfor col in categorial_features:\n    lbl = LabelEncoder()\n\n    df[col] = lbl.fit_transform(train[col].fillna('N/A').astype(str))\n    label_encoder_list.append(lbl)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"categorial_features[4], label_encoder_list[4].inverse_transform([5])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"onehot_encoder_list = []\nonehot_features_list = []\n\nfor col in categorial_features:\n    one = OneHotEncoder()\n    one_hot_form = one.fit_transform(df[col].values.reshape(-1,1))\n    onehot_features_list.append(one_hot_form)\n    onehot_encoder_list.append(one)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in onehot_features_list:\n    print(i.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ohehot_features = hstack(onehot_features_list).tocsr()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# np.stack","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ohehot_features","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def rmse(y_true, y_pred):\n    return (mean_squared_error(y_true, y_pred)**0.5).round(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"desccription_models = []\ntitle_models = []\nonehot_features_models = []\n\noof_predictions = np.zeros(shape=[train.shape[0], 3])\n\nfor fold_id, (train_idx, val_idx) in tqdm_notebook(enumerate(FOLD_LIST)):\n    \n    \n    descr_train, title_train, onehot_train, y_train = (\n        train_description_tfidf[train_idx],\n        train_title_tfidf[train_idx],\n        ohehot_features[train_idx],\n        Y[train_idx]\n    )\n\n    descr_val, title_val, onehot_val, y_val = (\n        train_description_tfidf[val_idx],\n        train_title_tfidf[val_idx],\n        ohehot_features[val_idx],\n        Y[val_idx]\n    )\n    \n    \n    descr_model = Ridge()\n    title_model = Ridge()\n    onehot_model = Ridge()\n    \n    descr_model.fit(descr_train, y_train)\n    oof_predictions[val_idx, 0] = descr_model.predict(descr_val)\n    desccription_models.append(descr_model)\n    \n    title_model.fit(title_train, y_train)\n    oof_predictions[val_idx, 1] = title_model.predict(title_val)\n    title_models.append(title_model)\n    \n    onehot_model.fit(onehot_train, y_train)\n    oof_predictions[val_idx, 2] = onehot_model.predict(onehot_val)\n    onehot_features_models.append(onehot_model)\n    \n    print('###', 'fold', fold_id,':', '###')\n    print('descr_model rmse:', rmse(oof_predictions[val_idx, 0], y_val))\n    print('title_model rmse:', rmse(oof_predictions[val_idx, 1], y_val))\n    print('onehot_model rmse:', rmse(oof_predictions[val_idx, 2], y_val))\n    print('#'*20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.DataFrame(np.array([oof_predictions for i in range(3)]).T)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rmse(np.zeros(1503424)+Y.mean(), Y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in range(3):\n    print(rmse(oof_predictions[:,i], Y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#     print('descr_model rmse:', rmse(oof_predictions[val_idx, 0], y_val))\n#     print('title_model rmse:', rmse(oof_predictions[val_idx, 1], y_val))\n#     print('onehot_model rmse:', rmse(oof_predictions[val_idx, 2], y_val))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.price.hist(bins=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.price.fillna(0).clip(0,5000000).hist(bins=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.log1p(train.price.fillna(0).clip(0,5000000)).hist(bins=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['log_clip_price'] = np.log1p(train.price.fillna(0).clip(0,5000000))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['log_clip_price']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[['parent_category_name','log_clip_price']].groupby(\n    'parent_category_name')['log_clip_price'].agg(['mean','max','std'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[['parent_category_name','log_clip_price']].groupby('parent_category_name')['log_clip_price'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[['parent_category_name','price']].groupby('parent_category_name')['price'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"agg_price = train[['parent_category_name','price']].groupby('parent_category_name')['price'].describe()\nagg_log_price = train[['parent_category_name','log_clip_price']].groupby('parent_category_name')['log_clip_price'].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mean_parent_category_price = agg_price['mean'].reset_index()\nprint(mean_parent_category_price.columns)\nmean_parent_category_price.columns = ['parent_category_name', 'mean_parent_category_price']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = train.merge(mean_parent_category_price)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"deviation_price = (train.price.fillna(0).clip(0,5000000) - train['mean_parent_category_price'])#.hist(bins=100)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"deviation_price.hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X = np.concatenate([\n    deviation_price.values.reshape(-1,1),\n    train['mean_parent_category_price'].values.reshape(-1,1),\n    oof_predictions,\n],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.DataFrame(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from catboost import CatBoostRegressor, Pool","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"_models = []\n\noof_predictions = np.zeros(shape=[train.shape[0]])\n\nfor fold_id, (train_idx, val_idx) in tqdm_notebook(enumerate(FOLD_LIST)):\n    \n    X_train, Y_train = X[train_idx], Y[train_idx]\n    X_val, Y_val = X[val_idx], Y[val_idx]\n    \n    eval_dataset = Pool(X_val, Y_val)\n    \n    model = CatBoostRegressor(\n        learning_rate = 0.1,\n        iterations=100, depth=16, max_leaves=37, eval_metric='RMSE',\n        metric_period=10, use_best_model=True,\n        grow_policy='Lossguide',\n        max_bin=1024\n        \n    )\n    model.fit(X_train, Y_train, eval_set = eval_dataset)\n    \n    _models.append(model)\n    preds = model.predict(X_val)\n    oof_predictions[val_idx] = preds\n    print('fold_id:', fold_id, rmse(Y_val, preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"rmse(oof_predictions, Y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# при идеальных результатах точки лежат на одной диагонали\nplt.figure(figsize=(10,10))\nplt.scatter(Y, oof_predictions, alpha=0.01, s=30)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.feature_importances_.round(1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# А теперь применим все эти преобразования к TEST датасету :)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}