{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd, numpy as np\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1175e08d3547e7fe2b16c7a1fc79cd0a12a480a2","collapsed":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv', usecols=['description', 'deal_probability'])\ntest = pd.read_csv('../input/test.csv', usecols=['description'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bf296c21e6dd96e5e69b3268e20c63f998345077","collapsed":true},"cell_type":"code","source":"COMMENT = 'description'\ntrain[COMMENT].fillna(\"неизвестный\", inplace=True)\ntest[COMMENT].fillna(\"неизвестный\", inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"5c56364babf4fbff764f5556b441a79126d0ba16"},"cell_type":"code","source":"import re, string\nre_tok = re.compile(f'([{string.punctuation}“”¨«»®´·º½¾¿¡§£₤‘’])')\ndef tokenize(s): return re_tok.sub(r' \\1 ', s).split()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"2e8e79fde86e1e1ca3e7ba04e5a18a15368c48d8"},"cell_type":"code","source":"n = train.shape[0]\nvec = TfidfVectorizer(ngram_range=(1,2), tokenizer=tokenize,\n               min_df=3, max_df=0.9, strip_accents='unicode', use_idf=1,\n               smooth_idf=1, sublinear_tf=1 )\ntrn_term_doc = vec.fit_transform(train[COMMENT])\ntest_term_doc = vec.transform(test[COMMENT])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e53f5a5f17d471410403beb5477d34eb81d08021","collapsed":true},"cell_type":"code","source":"trn_term_doc, test_term_doc\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"688baa23a7a6394c2be732d98562c1ad3a3b2454"},"cell_type":"code","source":"def pr(y_i, y, x_temp):\n    p = x_temp[y==y_i].sum(0)\n    return (p+1) / ((y==y_i).sum()+1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ca0daea486b798e66a2a20e121a643fc0985f122"},"cell_type":"code","source":"x = trn_term_doc\ntest_x = test_term_doc","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"dea606ce2d56e45cc3e0458d7abe496aa5fb23f0"},"cell_type":"code","source":"def get_mdl(y, x_temp):\n    y = y.values\n    r = np.log(pr(1,y, x_temp) / pr(0,y, x_temp))\n    m = linear_model.Lasso(alpha=0.1)\n    x_nb = x_temp.multiply(r)\n    return m.fit(x_nb, y), r","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb213368108bd9528fecf6b582ed86f20bef36a2","collapsed":true},"cell_type":"code","source":"from sklearn.model_selection import KFold\nfrom sklearn import metrics\nfrom sklearn import linear_model\nfrom sklearn.isotonic import IsotonicRegression\nRS = 20180601\nfolds = KFold(n_splits=5, shuffle=True, random_state=546789)\noof_preds = np.zeros(x.shape[0])\ntest_predicts_list = []\nnp.random.seed(RS)\nfor n_fold, (trn_idx, val_idx) in enumerate(folds.split(x)):\n    print(\"fold {}\".format(n_fold))\n    trn_x, trn_y = x[trn_idx], train.loc[trn_idx]\n    val_x, val_y = x[val_idx],  train.loc[val_idx]\n    m,r = get_mdl(trn_y['deal_probability'], trn_x)\n    oof = m.predict(val_x.multiply(r))\n    oof_preds[val_idx] = oof\n    print('RMSE:', np.sqrt(metrics.mean_squared_error(val_y['deal_probability'].values, oof)))\n    preds = m.predict(test_x.multiply(r))\n    test_predicts_list.append(preds)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"8377e10821ee8bfdb5f39cc9ceab8ee4a05e569f"},"cell_type":"code","source":"test_predicts = np.ones(test_predicts_list[0].shape)\nfor fold_predict in test_predicts_list:\n    test_predicts *= fold_predict\n\ntest_predicts **= (1. / len(test_predicts_list))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"ab5749dc5c5e4d398de36abd1494d74bdec06467"},"cell_type":"code","source":"np.save('lasso_naivebayes_oof.npy', oof_preds)\nnp.save('lasso_naivebayes_preds.npy', test_predicts)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"72416ec18f475c559e1cb7fa6b9ef73e64c8c91c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}