{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"# От такие у нас данные\ndf_train = pd.read_csv(\"../input/train.csv\")\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fbacdabf2a95ab1376e23a1ac1eab33449be412d"},"cell_type":"code","source":"# Заимпортим все шо нада\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import StratifiedKFold, GridSearchCV\nfrom sklearn.feature_extraction.text import TfidfVectorizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f21da061728de496fc912fe9aec610d6a2630ef5"},"cell_type":"code","source":"# Вот мы сделаем объект типа векторайзер\nvect = TfidfVectorizer()\n# А воть классификатор\nclf = SGDClassifier(loss = 'modified_huber')\n# А вот создали модель, которая будет вначале векторайзить, а потом классифицировать\nmodel = Pipeline([('vect', vect), ('clf', clf)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bcdd15090cabc733dbf5e42f09d01b2c557d158c"},"cell_type":"code","source":"# Посмотрим какие у нашей модели есть параметры\nmodel.get_params().keys()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"538690d391bc5929995a04713c6c6ae07f51d92b"},"cell_type":"code","source":"# Некоторые из них переберем, прям как в прошлых лабах\n# Перебор будет овер долгим, поэтому переберем чуть-чуть параметров\n\nmodel_grid = {\n    'vect__min_df': [0, 0.0001, 0.001],\n    'vect__max_df': [1.0, 0.9, 0.8],\n    'clf__tol': [1e-3, 1e-4, 1e-2]\n}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"897c36299d2b89f1962412980cca6b1ccac88430"},"cell_type":"code","source":"model_search = GridSearchCV(model, model_grid, cv=StratifiedKFold(4, random_state=3), \n                            n_jobs=4, verbose=True)\n\nX = df_train['question_text'].values\ny = df_train['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"042e49ce603f37c247b84e08042e486bae7a1ae1"},"cell_type":"code","source":"model_search.fit(X, y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"49c71dbd5b8062823726c2e80b266172554ffdfa"},"cell_type":"code","source":"# Посмотри на параметры лучшей модельки\nmodel_search.best_params_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a57a90b9847f6395b9a64927906735ee99753d48"},"cell_type":"code","source":"# Сохраним лучшую моделю\nbest_model = model_search.best_estimator_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d9c2f76cb737cdb7b2e1fe1dc2a4c4e2aff38f7"},"cell_type":"code","source":"%%time\n\n# Теперь сделаем предсказания, причем предсказывать будем вероятности, они нам интереснее\nmodel_preds = cross_val_predict(best_model, X, y,\n                          cv=StratifiedKFold(4, random_state=3), n_jobs=4,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"65c166243d57354b8842f5e0581b9d86e357d878"},"cell_type":"code","source":"# Переберем пороги для определения, плохой вопрос или норм. Таким образом найдел лучший порог\nfrom sklearn.metrics import f1_score\n\nthresholds = np.arange(0.05, 0.95, 0.05)\nopt_threshold = 0\nopt_score = 0\nfor threshold in thresholds:\n    tmp_pred = list(map(lambda x: 1 if x[1]>threshold else 0, model_preds))\n    tmp_score = f1_score(y, tmp_pred)\n    if tmp_score > opt_score:\n        opt_score = tmp_score\n        opt_threshold = threshold\n    print(f\"F1-score = {tmp_score} with threshold = {threshold}\")\n\nprint('----------------------------------------------------')\nprint(f\"Optimal threshold is {opt_threshold}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"359d17d8ded817e8c982ae1a2f003876aae1c522"},"cell_type":"code","source":"# Итак, теперь обучим моделю и сделаем предсказания на тестовых данных\n\nbest_model.fit(X, y)\ntest_data = pd.read_csv(\"../input/test.csv\")[\"question_text\"].values\ntest_id = pd.read_csv(\"../input/test.csv\")[\"qid\"].values\npredictions_prob = best_model.predict_proba(test_data)\nfinal_predictions = list(map(lambda x: 1 if x[1]>opt_threshold else 0, predictions_prob))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"651cac8e32dc15f5de2a8770918b98597d5f4dd7"},"cell_type":"code","source":"# Сделаем ответы в правильном виде\nanswers = pd.DataFrame(np.transpose([test_id, final_predictions]),\n                                   columns = [\"qid\", \"prediction\"])\nanswers.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"73f0164122256918232bb168b4159a335ddd7790"},"cell_type":"code","source":"answers.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}