{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import cross_val_score\nfrom lightgbm import LGBMRegressor\nfrom scipy import sparse\nfrom category_encoders.hashing import HashingEncoder\nimport os\nprint(os.listdir(\"../input\"))","execution_count":1,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\ntest = pd.read_csv('../input/test.csv')","execution_count":2,"outputs":[]},{"metadata":{"_uuid":"ec3ba18be4d9c486e75840bc237f401d65546b56","_cell_guid":"69dc84e6-4aa7-4845-80dc-4f6a2e47bde4","trusted":true},"cell_type":"code","source":"train.head()","execution_count":3,"outputs":[]},{"metadata":{"_uuid":"cb02fa504953f006dae9a6029f653862f16830c0","_cell_guid":"8681975c-97e8-44cc-a2c4-1fd2d93c6518","trusted":true},"cell_type":"code","source":"train.columns","execution_count":4,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"2d3e3d1a4b576ec82afc41652d2425750660353a","_cell_guid":"e277580c-3bb7-455c-bee2-73155cd4fc16","trusted":true},"cell_type":"code","source":"cat_feats = ['region', 'city', 'parent_category_name', 'category_name', 'param_1', 'param_2', 'param_3', 'user_type', 'image_top_1']\ntext_feats = ['title', 'description']\nnum_feats = ['price', 'item_seq_number']\nallcols = cat_feats + text_feats + num_feats","execution_count":5,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"ffa4e7a362d60a4d0655c20074ff55f63d4d13ed","_cell_guid":"64ed3355-d4fb-423c-b9c5-f729d6c7719d","trusted":true},"cell_type":"code","source":"merged = pd.concat((train[allcols], test[allcols]), axis=0)\nmerged['price'] = merged['price'].apply(np.log1p)","execution_count":6,"outputs":[]},{"metadata":{"_uuid":"c73999ffe23c647bb23c19fbf972101274ea7498","_cell_guid":"38233351-d449-478e-9574-ea2635f69d78"},"cell_type":"markdown","source":"### Text Features TfIdf"},{"metadata":{"_uuid":"ca7909b5ef5add3cb482f8b0bf38df89eb76e631","_cell_guid":"d09f9a7b-3beb-4969-9e91-cd1bae5eb24e","trusted":true,"collapsed":true},"cell_type":"code","source":"import pymorphy2\nimport re\n\nmorph = pymorphy2.MorphAnalyzer()\nretoken = re.compile(r'[\\'\\w\\-]+')","execution_count":7,"outputs":[]},{"metadata":{"_uuid":"4e1ec9ce4528aae44883a0ee86ea02a59fcf0122","_cell_guid":"b76d7ace-7db6-446f-8f5a-cf6ae6ef1df9"},"cell_type":"markdown","source":"### Why normalize\nRussian language has inflectional structure, so same word is written different ways in different context.  \nFor example:  \nDog -> Собак**а**  \nNo dog -> нет собак**и**  \nGive a dog a bone -> Дай собак**е** кость.  \nEtc. There are also many more complicated cases. Here is example how to normalize russian text. It should decrease dictionary and increase Tfidf quality."},{"metadata":{"_uuid":"d6865834767c469abfce1e2edff4eab9e4a6e313"},"cell_type":"markdown","source":"##### Here is example on competition's data"},{"metadata":{"_cell_guid":"dab1311c-a7aa-4ba6-a72e-a254d85bf0e5","_uuid":"81081be6ed473bd5a7dd2d7dffa97d6a8d89e4f5","trusted":true},"cell_type":"code","source":"s = merged['description'].tail().values[-1]\nprint(s)","execution_count":8,"outputs":[]},{"metadata":{"_cell_guid":"d10c965a-ed9c-4804-b032-9bce79612e78","_uuid":"dd2d5c02f709af0172514b2ed1afbeedde1fb6e3","collapsed":true,"trusted":true},"cell_type":"code","source":"def tokenize_normalize(text):\n    text = retoken.findall(text.lower())\n    text = [morph.parse(x)[0].normal_form for x in text]\n    return ' '.join(text)","execution_count":9,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f589cdd66979cc9c368c8489336cb1356f2b9e19"},"cell_type":"code","source":"tokenize_normalize(s)","execution_count":10,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"fb2d002bf4a11fc60afa2a82d3bbac8026e4cf58"},"cell_type":"code","source":"# some descriptions only consist of a digits\nmerged['description'] = merged['description'].astype(str)","execution_count":11,"outputs":[]},{"metadata":{"_uuid":"4700beaa62e136c19b2ad23c574479ba3ede4c6e"},"cell_type":"markdown","source":"### Normalize:"},{"metadata":{"_cell_guid":"c5fc4fa8-03c1-4dec-b48e-8f18d7b3218d","_uuid":"3abe1710313f18b525ab80b7a1b1bd19b3e898b7","collapsed":true,"trusted":true},"cell_type":"code","source":"%%time\nmerged['description_norm'] = merged['description'].apply(tokenize_normalize)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b1e529af0136a1179f4ccf4ab52305fa95d96d17","collapsed":true,"_cell_guid":"31324333-5731-42a0-8fb0-72abeeea34f0","trusted":true},"cell_type":"code","source":"%%time\ntfidf = TfidfVectorizer(ngram_range=(1, 3), encoding='KOI8-R', min_df=100, max_df=0.999)\ntfidf_matrices = []\nfor feat in ['description_norm', 'title']:\n    tfidf_matrices.append(tfidf.fit_transform(merged[feat].fillna('').values))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6928faf20a6f6d800c3a8183f7d846355ed7cba1","collapsed":true,"_cell_guid":"d8739396-1eef-46f3-bd53-b7b16dc1cc8c","trusted":true},"cell_type":"code","source":"%%time\ntfidf_matrices = sparse.hstack(tfidf_matrices, format='csr')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"3627ba7e8282058059e6e44e679848b5dc90459b"},"cell_type":"code","source":"print(tfidf_matrices.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"789a7c48c0cf56f38b85d83286cc9b82923171d9","_cell_guid":"10b24f1d-d3e8-45d1-b654-a764ff5d5e19"},"cell_type":"markdown","source":"### Categorical Features Hashing"},{"metadata":{"_uuid":"364cc8f6569b72e09d0892424801cd9eff942275","collapsed":true,"_cell_guid":"feda058d-348b-4dae-8cc4-275e4247519d","trusted":true},"cell_type":"code","source":"%%time\nhe = HashingEncoder()\ncat_df = he.fit_transform(merged[cat_feats].values)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"aa9f9d0eee2b6d7fc7440052895eccc18f377427","collapsed":true,"_cell_guid":"66af7f37-f2f3-4954-b69b-73258ffcf342","trusted":true},"cell_type":"code","source":"cat_df.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"61166b53caea0e55ef59096a49d0e2237980f2ba","_cell_guid":"1ae307a3-adcf-4a72-a0b0-96e32699fe3c"},"cell_type":"markdown","source":"### All Features together + CV"},{"metadata":{"collapsed":true,"_uuid":"6a8af815ea77abc7baa1bd97eae2106fa21b5585","_cell_guid":"201ce4e1-4127-4c9e-a3d1-079540a04ce4","trusted":true},"cell_type":"code","source":"full_matrix = sparse.hstack([cat_df.values, tfidf_matrices, merged[num_feats].fillna(-1).values], format='csr')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"86ae661295af9ed3f4b265312614ba60e4b79701","collapsed":true,"_cell_guid":"aa8edc87-661f-4d2e-883a-f4edfddf11f0","trusted":true},"cell_type":"code","source":"import gc\ndel tfidf_matrices, merged, cat_df\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5a956044dcccf32ef98d405a7a160c841b33270a","collapsed":true,"_cell_guid":"7c657965-ae43-4046-bf74-38695134aba4","trusted":false},"cell_type":"code","source":"%%time\nmodel = LGBMRegressor(max_depth=4, learning_rate=0.3, n_estimators=550)\nres = cross_val_score(model, full_matrix[:train.shape[0]], train['deal_probability'].values, cv=4, scoring='neg_mean_squared_error')\nres = [np.sqrt(-r) for r in res]\nprint(np.mean(res), np.std(res))","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"696ec11927cf31232b8a139022234717604abeed","_cell_guid":"2d7e87f1-1122-4f51-b07a-aafbaa32ff6e","trusted":false},"cell_type":"code","source":"model.fit(full_matrix[:train.shape[0]], train['deal_probability'].values)\npreds = model.predict(full_matrix[train.shape[0]:])","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"b0fed347f37d992d3b76feba1e9553f8efa10fdf","_cell_guid":"dd4c09c0-2bce-4a0d-a74a-5f645de4dc9b","trusted":false},"cell_type":"code","source":"import matplotlib.pyplot as plt\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f429043049fc2758d6f65cced968fa411903aefd","collapsed":true,"_cell_guid":"6059501c-c1cb-40a0-9a35-0db72a4b2545","trusted":false},"cell_type":"code","source":"plt.figure(figsize=(10, 7))\nplt.hist(preds, bins=50);","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"_uuid":"ae0a891499aa89e7fff4be64d3b51ad01b73bcf3","collapsed":true,"_cell_guid":"929866c4-7274-4a27-8053-11691a650f9a","trusted":false},"cell_type":"code","source":"sub = pd.read_csv('../input/sample_submission.csv')\nsub['deal_probability'] = preds\nsub['deal_probability'].clip(0.0, 1.0, inplace=True)\nsub.to_csv('first_attempt.csv', index=False)\nsub.head()","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}