{"cells":[{"metadata":{"_cell_guid":"9420df04-9fea-48f9-8170-e8d769bef077","_uuid":"c4c42567fce3025b11bdb865b93efbaedf213bd3"},"cell_type":"markdown","source":"This kernel is based on work from Beep-beep: https://www.kaggle.com/the1owl/beep-beep"},{"metadata":{"collapsed":true,"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"from nltk.corpus import stopwords\nfrom datetime import datetime\nimport lightgbm as lgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nimport pandas as pd\nimport numpy as np\nimport gc\nfrom scipy import sparse\nimport gzip\nfrom sklearn.decomposition import TruncatedSVD\nfrom pathlib import PurePath\nimport matplotlib.pyplot as plt\nfrom sklearn.svm import LinearSVR\nfrom scipy import sparse\n\n%matplotlib inline","execution_count":44,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"42370b1a-f719-4d4f-9841-aaa862963c97","_uuid":"a0bb249854185d939ac7e7df0cdcfe77514a9d43","trusted":true},"cell_type":"code","source":"target_col ='deal_probability'\ntoy = False # Activate for debug proposes\nvalidate = True\nbest_num_boost_round = 100 if toy else 5000\nnum_boost_round = 100 if toy else 5000","execution_count":45,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"%%time\ndtypes = {\n    'user_id': 'category',\n    'region': 'category',\n    'city': 'category',\n    'parent_category_name': 'category',\n    'category_name': 'category',\n    'param_1': 'category',\n    'param_2': 'category',\n    'param_3': 'category',\n    'title': 'str',\n    'description': 'str',\n    'price': 'float',\n    'item_seq_number': 'int',\n    'activation_date': 'object',\n    'user_type': 'category',\n    'image': 'str',\n    'image_top_1': 'float',\n    'deal_probability': 'float'\n}\ndate_cols = ['activation_date']\n\n# Replace category by 'object' for easier join of train and test\ndtypes_load = {k:('object' if v == 'category' else v) for k, v in dtypes.items()}\n\ndf_train = pd.read_csv('../input/avito-demand-prediction/train.csv', dtype=dtypes_load, parse_dates=date_cols, index_col=\"item_id\", nrows=100000 if toy else None)\ndf_test = pd.read_csv('../input/avito-demand-prediction/test.csv', dtype=dtypes_load, parse_dates=date_cols, index_col=\"item_id\", nrows=100000 if toy else None)\n\nn_train = df_train.shape[0]","execution_count":46,"outputs":[]},{"metadata":{"_cell_guid":"cc60b337-8a24-4357-94f5-845dca97cb65","_uuid":"99afea99822dbae48b22189aa6e252a33d6f09eb"},"cell_type":"markdown","source":"## Image features"},{"metadata":{"collapsed":true,"_cell_guid":"cf5390ce-63a8-4f8b-bedb-b8c136349e24","_uuid":"3d2ce23942b8bdfa007a2ffae45d976a6c8e1b38","trusted":true},"cell_type":"code","source":"def load_imfeatures(folder):\n    path = PurePath(folder)\n    features = sparse.load_npz(str(path / 'features.npz'))\n    \n    if toy:\n        features = features[:100000]\n        \n    return features","execution_count":47,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"94884bfb-1da6-49c2-86f5-b9657c8dbe00","_uuid":"e3a21559dd28f155cbef54d3b3408f3c1fb4ad8c","trusted":true},"cell_type":"code","source":"ftrain = load_imfeatures('../input/vgg16-train-features/')","execution_count":48,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"1eedd160-8e48-4725-b86e-b55afd1b0931","_uuid":"71e5c0134bdde75dd16cbfe6a8e1b6782157c753","trusted":true},"cell_type":"code","source":"ftest = load_imfeatures('../input/vgg16-test-features/')","execution_count":49,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"febdc1be-8e3c-434b-8a1b-8dddb39e5704","_uuid":"3e875df5ad078d8ca2dd21c7358e8fbd9f81e973","trusted":true},"cell_type":"code","source":"assert df_train.shape[0] == ftrain.shape[0]\nassert df_test.shape[0] == ftest.shape[0]","execution_count":50,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"7b6de579-5d7b-403a-a56d-c7319c1bd66c","_uuid":"eddcc4766e006dd83a741c86fa746f8fd69aafc6","trusted":true},"cell_type":"code","source":"# Create both dataframe\ndf_target = df_train[target_col]\ndf_both = pd.concat([df_train, df_test])\n\ndel df_train, df_test\ngc.collect();","execution_count":51,"outputs":[]},{"metadata":{"_cell_guid":"7d744ecd-956b-4b81-8046-98a1dc8d85bc","_uuid":"a134638f3f08f635137f334bb82484a7dcd43d94","trusted":true},"cell_type":"code","source":"fboth = sparse.vstack([ftrain, ftest])\ndel ftrain, ftest\ngc.collect()\nfboth.shape","execution_count":52,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"da56cea7-8e59-49ac-ace7-9b7aa40a0ca9","_uuid":"4aa8cb0616a323ce1b771c1242708c294f47c4a8","trusted":true},"cell_type":"code","source":"# Categorical image feature (max and min VGG16 feature)\ndf_both['im_max_feature'] = fboth.argmax(axis=1)  # This will be categorical\ndf_both['im_min_feature'] = fboth.argmin(axis=1)  # This will be categorical\n\ndf_both['im_n_features'] = fboth.getnnz(axis=1)\ndf_both['im_mean_features'] = fboth.mean(axis=1)\ndf_both['im_meansquare_features'] = fboth.power(2).mean(axis=1)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"168f25e6-2537-4fd5-847d-81d300222f87","_uuid":"0fe0242ce37c32d96070afc60828766e83f1e3c9","trusted":true},"cell_type":"code","source":"%%time\n# Let`s reduce 512 VGG16 featues into 32\ntsvd = TruncatedSVD(32)\nftsvd = tsvd.fit_transform(fboth)\ndel fboth\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"7225c3c4-d9c8-4fb3-b06f-6522143b24c9","_uuid":"b3dfe3b041d193f2468b6143510a30c8f23b9b9d"},"cell_type":"markdown","source":"## Feature engineering"},{"metadata":{"collapsed":true,"_cell_guid":"15ad9975-6d8b-4f88-8c7d-800c4ec1407c","_uuid":"918473ff8d5bf9faa789962cd0954354c660d643","trusted":true},"cell_type":"code","source":"# Convert df_both categorical cols to 'category' type\nfor col, dtype in dtypes.items():\n    if dtype == 'category':\n        df_both[col] = df_both[col].astype('category').cat.codes","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"d9601bd0-8dfd-4145-aab8-4883943e345e","_uuid":"8aac03e9fdc12d5c39dd791a57b9249e5ddd959b","trusted":true},"cell_type":"code","source":"df_both.dtypes","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"e7e49ad9-c9cb-44f2-bd8c-7a912116ac19","_uuid":"ab7cb08e27ec58a369e07b77e0cc6efb65eecc88","trusted":true},"cell_type":"code","source":"cat_cols = ['region', 'city', 'parent_category_name', 'category_name', 'param_1', 'param_2', 'param_3', 'user_type',\n           'im_max_feature', 'im_min_feature']\nnum_cols = ['price', 'image_top_1', 'deal_probability']\n\nfor cat_col in cat_cols:\n    df_group = df_both.groupby(cat_col)[num_cols].agg(['mean', 'std'])\n    df_group.columns = ['{}_'.format(cat_col) + '-'.join(col).strip() for col in df_group.columns.values]\n\n    df_both = df_both.join(df_group, on=cat_col, how='left')\n    del df_group\n    print(cat_col)\n    \n# del df_train\n# gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"41748e85-1cd6-4b8d-979f-e54a60b85182","_uuid":"a21f4ab589f353497ff9c339627ef757f1ce6e36","trusted":true},"cell_type":"code","source":"df_both['activation_dow'] = df_both['activation_date'].dt.dayofweek","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"82c5376e-359f-46ef-9e28-80a167d7215f","_uuid":"7303c002f3ec829c70ff9847ce39701dc6acbdff","trusted":true},"cell_type":"code","source":"# Text processing\nsw = stopwords.words('russian')\ntxt_cols = ['title', 'description']\n\nfor txt_col in txt_cols:\n    df_both[txt_col + '_len'] = df_both[txt_col].str.len()\n    df_both[txt_col + '_wc'] = df_both[txt_col].str.count(' ')\n    \n    if txt_col != 'description':\n        feature_cnt = 50\n\n        tfidf = TfidfVectorizer(stop_words=sw, min_df=10, max_df=0.8, dtype=np.float32, max_features=32000)\n        X_text = tfidf.fit_transform(df_both[txt_col])\n        svr = LinearSVR(C=0.01).fit(X_text[:n_train], df_target)\n        fnames = sorted(tfidf.vocabulary_, key=tfidf.vocabulary_.__getitem__)\n        \n        best_features = np.argsort(svr.coef_)\n        \n        # Select most positive and negative features\n        selected_features = np.concatenate([best_features[:feature_cnt], best_features[-feature_cnt:]])\n        features_names = list(map(fnames.__getitem__, selected_features))\n        \n        df_both_tfidf = pd.DataFrame(X_text[:, selected_features].todense(), columns=features_names, index=df_both.index)\n        \n        del X_text, features_names, selected_features, best_features, svr, fnames, tfidf\n\n        df_both = df_both.join(df_both_tfidf)\n        \n        del df_both_tfidf\n        gc.collect()\n        \n    print(txt_col)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"e10de7b0-0764-421f-917f-02337f391076","_uuid":"92a3892d7b4b005b8116155461b33a5d1cf17732","trusted":true},"cell_type":"code","source":"# Merge image features into df_both\ndf_ftsvd = pd.DataFrame(ftsvd, index=df_both.index).add_prefix('im_tsvd_')\n\ndf_both = pd.concat([df_both, df_ftsvd], axis=1)\n\ndel df_ftsvd, ftsvd\ngc.collect();","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ac2acacb-4c44-4122-a13d-6a4b7d3faf56","_uuid":"49616e5bf18115173547ffaca8d034cfc845cdc9","trusted":true,"collapsed":true},"cell_type":"code","source":"# Split df_both in train and test\ndf_train = df_both.iloc[:n_train]\ndf_test = df_both.iloc[n_train:]\n\ndel df_both\ngc.collect()\n\ndf_train.shape, df_test.shape","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"1d3d3154-4ac5-4377-906f-01579c5716e1","_uuid":"e741cb77f3e7006354043b2fa1a8ea0e4225b1c4","trusted":true,"collapsed":true},"cell_type":"code","source":"ex_cols = {'item_id', 'user_id', 'deal_probability', 'title', 'description', 'image', 'activation_date'}\nused_cols = [c for c in df_train.columns if c not in ex_cols]\nprint('Used cols:', ', '.join(used_cols))","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"ceb68014-f533-4f78-96ec-e96dff3eb176","_uuid":"00fef846cc3b51427fe805932e4590276477e235","trusted":true},"cell_type":"code","source":"if validate:\n    fold_train, fold_valid, target_train, target_valid = train_test_split(df_train[used_cols], df_target, test_size=0.2, random_state=42)\n\n    dtrain = lgb.Dataset(fold_train, target_train, categorical_feature=cat_cols)\n    dvalid = lgb.Dataset(fold_valid, target_valid, categorical_feature=cat_cols)\n    \n    valid_sets = [dvalid]\n    valid_names = ['valid']\n    \n    del fold_train, fold_valid, target_train, target_valid\nelse:\n    dtrain = lgb.Dataset(df_train[used_cols], df_target)\n    valid_sets = [dtrain]\n    valid_names = ['train']\n    assert best_num_boost_round is not None","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"dd1a4e3e-a598-4923-9272-d26174956e28","_uuid":"9797657f67220bb9dc81531b29595b5e032c107d","trusted":true,"collapsed":true},"cell_type":"code","source":"%%time\n# LGB train\nparams = {\n    'learning_rate': 0.02,\n    'boosting': 'gbdt',\n    'objective': 'regression',\n    'metric': ['rmse'],\n    'is_training_metric': True,\n    'seed': 19,\n    'num_leaves': 31,\n    'feature_fraction': 0.9,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 5\n}\n\nmodel = lgb.train(params, dtrain, num_boost_round=num_boost_round, valid_sets=valid_sets, valid_names=valid_names,\n                  verbose_eval=num_boost_round//20, early_stopping_rounds=50 if validate else None)\n\nif validate:\n    best_num_boost_round = model.best_iteration","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"aac86d0c-e519-41dc-95aa-ce305a52bc9f","_uuid":"ce5060a62ad3c173e8c590c012720b5f50317aee","trusted":true,"collapsed":true},"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(8, 25))\nlgb.plot_importance(model, ax=ax);","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"954d870d-8854-49ab-b4a4-302ee1f2cc1e","_uuid":"1cb3b5f36e633028d368207385ae773f3aec9996","trusted":true,"collapsed":true},"cell_type":"code","source":"print(best_num_boost_round)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"ec3a3aa5-dd99-4649-9dc1-1f67156aa5fb","_uuid":"376a825a3f2c5bc6c857e06b1a938e289bb8e350","trusted":true,"collapsed":true},"cell_type":"code","source":"del df_train, dtrain\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_cell_guid":"75b34402-bb90-4a81-a38f-734cd4dea1f7","_uuid":"a44f084e670ffd058e528de3edcc714b2121e62c","trusted":true},"cell_type":"code","source":"df_test.index.name = 'item_id'\ndf_test['deal_probability'] = model.predict(df_test[used_cols], num_iteration=best_num_boost_round).clip(0., 1.)\ndf_test[['deal_probability']].to_csv('submission.csv', index=True)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"c7c934d6-19af-470c-91ca-d6c699800357","_uuid":"0abd3790dd840a3094895c8386b93e77053ed02a","trusted":true,"collapsed":true},"cell_type":"code","source":"!head submission.csv","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"3a241c21-ca99-4671-a523-89d2911f14c4","_uuid":"564ae85901172c4c1cdd0aab8b990d75d7eed016","trusted":true,"collapsed":true},"cell_type":"code","source":"df_test.shape","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}