{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df=pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5d89708ddde966cba21fd634cd53a462d5cb9689"},"cell_type":"code","source":"import eli5\nimport matplotlib.pylab as plt\n#загрузка дополнительных данных для nltk\nimport nltk\nnltk.download('punkt')\nnltk.download('stopwords')\nnltk.download('wordnet')\nfrom nltk.tokenize import word_tokenize\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom nltk.corpus import stopwords\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score, classification_report, f1_score\nfrom sklearn.model_selection import RandomizedSearchCV\nfrom nltk.stem import SnowballStemmer, WordNetLemmatizer, LancasterStemmer\nfrom functools import lru_cache","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f477495799196a55ded7569785de3c9cc2e35cf"},"cell_type":"markdown","source":"# Часть 0. Ваше впечатление от датасета"},{"metadata":{"_uuid":"59cf20492ba03ef393de9a49722ae31ce9011aab"},"cell_type":"markdown","source":"Сделайте аналитику на ваш выбор. Посмотрите на распределение целевой переменной, почитайте тексты, для которых она равна 1. Сделайте выводы"},{"metadata":{"trusted":true,"_uuid":"faa754a68d26cb656a29abb1621141e6641bdd78","scrolled":true},"cell_type":"code","source":"df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3f19e86e76e0e6e78c010405aa13d0ffd1ac055"},"cell_type":"code","source":"len(df.target[df.target==1]), len(df.target)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a25dbc0cabf9efe8dbabd02635c7d95c960596d8"},"cell_type":"code","source":"small_df=df\nsmall_df[small_df.target==0][1:20:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42b506bfaefb072e32c4b7757202804ecd8c1a3d"},"cell_type":"code","source":"len(df[df.target==0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dc44b996ddbd64b59f5b930ff1c139c17826b140"},"cell_type":"code","source":"survivals = df[df.target ==0][0:1225312:200]\nlen(survivals)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f4cbdae3e75266c4538ae1cec46806215db48c3b"},"cell_type":"code","source":"survivals.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20bb3444d995cbf72297a8949a1d141a4f2f4941"},"cell_type":"code","source":"df_small=pd.concat([survivals,df[df.target==1]])\nlen(df_small)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e195243496e7cf51d939fa5ae622f2cb4fbeae9e"},"cell_type":"code","source":"from sklearn.utils import shuffle\ndf_small = shuffle(df_small).reset_index()\nlen(df_small)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d10d1fe49ee9c82e2367d4af50f005a712797887"},"cell_type":"code","source":"df_small=df_small.drop('index',axis=1)\nlen(df_small)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67d238a42cdec3527014c4283ac16e81a1b27c03"},"cell_type":"code","source":"df_small.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e853cdd0ff3e11aac59ae3137bf92e6cb4e5f304"},"cell_type":"code","source":"df_small.question_text[7]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"04e40155d00c200706ccca5a5faff390e09d7c54"},"cell_type":"code","source":"from sklearn.svm import OneClassSVM","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"108a20788b43cbc7ba1c20b47b22080fbb13d99c"},"cell_type":"code","source":"vect = TfidfVectorizer()\nsvm = OneClassSVM()\nmodel = Pipeline([('vect', vect),('svm', svm)])\n\n#%%time\n#preds = cross_val_predict(model, df.question_text.values, df.target.values,\n #                         cv=StratifiedKFold(4), n_jobs=1,\n  #                        method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a342fbbfacbca923008eced079e235b7866b036"},"cell_type":"code","source":"model.fit(small_df.question_text)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"14ad3762bd7ad633214e97f23219f5f4dcc8dfe3"},"cell_type":"markdown","source":"пропусков нет, \n\n\"плохих\" вопросов оч мало :с"},{"metadata":{"trusted":true,"_uuid":"8de9856b863c95b929d61eac601008f1d93edebc"},"cell_type":"code","source":"plt.style.use('ggplot')\n\nwordnum = df['question_text'].apply(lambda x: len(str(x).split(\" \")))\nletternum = np.clip(df['question_text'].str.len(),0,300)\nplt.figure(figsize=(16,10))\nplt.subplot(2,2,1)\nplt.hist(letternum.loc[df.target==1],bins=20)\nplt.subplot(2,2,2)\nplt.hist(letternum,bins=20) \nplt.subplot(2,2,3)\nplt.hist(wordnum.loc[df.target==1],bins=20)\nplt.subplot(2,2,4)\nplt.hist(wordnum,bins=20) \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"56f449dd1733e7c465c8b695224be5dbdc1269ce"},"cell_type":"markdown","source":"# Часть 1. Tf-Idf"},{"metadata":{"_uuid":"7d2139f685226016b8f23453291eee819f933432"},"cell_type":"markdown","source":"В этой части необходимо пройти классический путь решения задачи методом \"мешок слов\". Ваша задача:\n1. Нормализовать данные (если нужно)\n2. Выбрать вариант токенизации (начните с from nltk import word_tokenize и подумайте, достаточно ли его)\n3. Определитесь со списком стоп-слов.\n4. Постройте конвейер (Pipeline), включающий в себя превращение текстов в векторы и непосредственно моделирование\n5. Объясните предсказания вашей модели при помощи eli5\n5. Попробуйте подобрать лучшие параметры при помощи GridSearchСМ или RandomizedSearchCV. Второй вариант НАМНОГО лучше, но нужно разобраться.\n6. Попробуйте как можно сильнее сжать пространство признаков путем фильтрации редких слов, при этом старайтесь не потерять в качестве.\n\nНиже по пунктам подробнее. Не надо волноваться, если сразу что-то будет не получаться. На практике в среду будут ответы на все ВОПРОСЫ по поводу ВСЕХ непонятных моментов. Если вопросов не будет, то и объяснений не будет, а во время сдачи лабы  фразы вида \"было непонятно\" будут игнорироваться."},{"metadata":{"_uuid":"c789053238f7f71e53d021346669e1482b071c77"},"cell_type":"markdown","source":"# Задание 1.1 Нормализация и токенизация"},{"metadata":{"_uuid":"bb1e547e029292cb9c28c4c7fdbe1d34b019d8e2"},"cell_type":"markdown","source":"Стоит быть честным и сказать, что тексты вопросов написаны очень хорошо, без шума, разделены в основном пробелами. Поэтому эта часть тривиальна. Поэтому, просто пощупайте и сравните 2 токенизатора."},{"metadata":{"trusted":true,"_uuid":"fc3c3cd3a8be06dfe5fc13dbe2d286c213280220"},"cell_type":"code","source":"word_tokenize(df.question_text[1],language='english')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"37f47c5fbb456a67a02331f97dd7f52ef888cb33"},"cell_type":"code","source":"vecc = TfidfVectorizer()\nsklear_tokenizer = vecc.build_tokenizer()\nsklear_tokenizer(df.question_text[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05bce4c9ce87d3c22086f126ba4e9539cd50b980"},"cell_type":"code","source":"# этот сам выбросил знаки препинания, ето приятно","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"35d3a714a3f6e8382aa5c5ddd6e5ff46a8947081"},"cell_type":"code","source":"X = vecc.fit_transform(df.question_text[1:3])\nvecc.get_feature_names()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"223742ba2690e1fa76d066b2088b25c1b1da3e7c"},"cell_type":"code","source":"df.question_text[1:3]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"84d1980d465114bac1ce3938ccf333bc2ed7c43e"},"cell_type":"code","source":"print(X)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d6ec93612aefa89c11e4436cb8588fb54bd34456"},"cell_type":"markdown","source":"# Часть 1.2 Стоп-слова"},{"metadata":{"_uuid":"585c2c603c798e5037b1ca459859845d2bd23a1f"},"cell_type":"markdown","source":"Токенизация должна быть согласована со стоп-словами. Т.е. токенизатор должен воспринимать стоп-слова как токены, а не как несколько токенов. В этом задании вам нужно посмотреть на стоп-слова из тдел и проверить, являются ли два токенизатора выше согласованными со списком английских стоп-слов из nltk. Для этого попробуйте токенизировать по очереди все стоп-слова двумя текенизаторами и сделайте соответствующие выводы."},{"metadata":{"trusted":true,"_uuid":"16ef813c84f3e246be0bb686f5d3e3a3e2434567"},"cell_type":"code","source":"from nltk.corpus import stopwords","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c2bb31f1cf549ec844574b144b1245c280a2d208"},"cell_type":"code","source":"vecc = TfidfVectorizer(stop_words = 'english')\nvecc.fit(df.question_text.values)\nsklear_tokenizer = vecc.build_tokenizer()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1f61bfe332f602a5f5bb65cfbf5e0372a8e6076c"},"cell_type":"code","source":"stopwords.words('english')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed958ec834a955d95dd622642a2854ca171ad7cd"},"cell_type":"code","source":"sklear_tokenizer('you have a dog, I can swim')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a35c05bda29b5fce49985d1b866475db123a267"},"cell_type":"code","source":"s1 = []\nfor w in stopwords.words('english'):\n    if len(sklearn_tokenizer(w))==0: s1.append(w)\ns1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"73e2e6e57de7a3555d58d60012706372c74ee77c"},"cell_type":"code","source":"sklearn_tokenizer('q w e r t y u i o p a s d f g h j k l z x c v b n m')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ed114727ba747ecc95c230154ec7a1176fd59f9"},"cell_type":"code","source":"vect = TfidfVectorizer(min_df=0.001,max_df=0.99,stop_words='english')\nvect.fit(df.question_text.values)\n# ладно, похоже, это не лечится\nsklearn_tokenizer = vect.build_tokenizer()\ns1 = []\nfor w in stopwords.words('english'):\n    if len(sklearn_tokenizer(w))==0: s1.append(w)\ns1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"52ae5dd117e7530cd7d0863ff6cc20705b323a1b"},"cell_type":"code","source":"s1 = []\nfor w in stopwords.words('english'):\n    if len(word_tokenize(w))==0: s1.append(w)\ns1\n# даже не старался","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c2a0bc03ccb1e85167d18d9cf959d7bda733d446"},"cell_type":"markdown","source":"# Часть 1.3 Pipeline и процесс обучения"},{"metadata":{"_uuid":"90c94cd37294a972fc43be31e36383124e89f318"},"cell_type":"markdown","source":"Когда в процессе моделирования есть несколько шагов, их бывает удобно объединить в конвейер. В sklearn это реализовано в классе Pipeline. Ниже вам представлен пример его использования.\nТакже в этой лабе мы будем пользоваться cross_val_predict, чтобы считать честные предсказанные вероятности. Для этого указан параметр method='predict_proba' в примере внизу. Как вы можете видеть, векторизатор и классификатор остались с параметрами по умолчанию. Ваша цель - **сделать для них нормальные параметры и полностью разбираться в том, что они значат** на основе знаний, которые вы получили на всех прошлых занятиях. На практике у вас обязательно будут спрашивать про эти параметры, а также про то, как работает TfIdfVectorizer и SGDClassifier. Типичные вопросы, ответы на которые необходимо знать для хорошей оценки:\n1. Что такое стохастический градиентный спуск?\n2. Как обучается и делает предсказания логистическая регрессия?\n3. Что такое tf-idf и зачем это нужно?\n4. Что такое кросс-валидация? Как производится кросс-валидация классом StratifiedKFold?\n\nИ, конечно же, многие другие вопросы. Аналогичные вопросы будут и к другим пунктам лабы.\n\n"},{"metadata":{"_uuid":"14989d3ffbc75692f602aa2c904a991aacd31469"},"cell_type":"markdown","source":"В соревновании результаты ранжируются по метрике F1 (обязательно спрошу ее формулу!!). Но мы локально будет считать также roc_auc_score и делать classification_report (Не забудьте указать либо подобрать решающую границу перед подачей предсказаний в эту функцию)"},{"metadata":{"trusted":true,"_uuid":"f4f48ec5ff929231625a04cc37094b17e4a080a2"},"cell_type":"code","source":"from sklearn.decomposition import TruncatedSVD","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0f2c40f61a111dc03c79958f17602c17c6dca471"},"cell_type":"markdown","source":"штука по умолчанию (почти по умолчанию, т.к. стоящая по умолчанию loss='hinge' не умеет считать вероятности)"},{"metadata":{"trusted":true,"_uuid":"254c02d70790221cf83780529bef675f93815028"},"cell_type":"code","source":"vect = TfidfVectorizer()\nclf = SGDClassifier(loss='modified_huber')\n#svd = TruncatedSVD(n_components=100)\nmodel = Pipeline([('vect', vect),('clf', clf)])\n\n#%%time\npreds = cross_val_predict(model, df.question_text.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f396d3a12bca9223871f26fe083e9f5b4e986606"},"cell_type":"code","source":"# подбираем границу\nfor i in np.arange(0.1,0.8,0.1):\n    print(\" {0} for bound {1}\".format(f1_score(df.target.values,preds[:,1]>=i),i))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b0fcb04a725ea70dfe8c0fe9e3001c70feaf9d8"},"cell_type":"code","source":"for i in np.arange(0.9,0.2,-0.1):\n    print(\"{0} for bound {1}\".format(f1_score(1-df.target.values,preds[:,0]>=i),i))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cd2e0eee5cf95b55c0640bfca319f6c8493aabb9"},"cell_type":"code","source":"print(classification_report(df.target.values,preds[:,1]>=0.3)), roc_auc_score(df.target.values,preds[:,1]), f1_score(df.target.values,preds[:,1]>=0.3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"34d5c9e9b4c28b1793387b6c24225bda190cd16a"},"cell_type":"markdown","source":"> много слов, возможно не очень-то и нужных\n> некоторые из них вообще ничего определенного не значат"},{"metadata":{"trusted":true,"_uuid":"d51f76fe98dbe2f5ee54234e865881be0188d4dc"},"cell_type":"code","source":"model.fit(df.question_text.values, df.target.values)\neli5.show_weights(model, top=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40379446ac7097904292293e2d1fcc7af368b78b"},"cell_type":"code","source":"#vect = TfidfVectorizer(min_df=0.0005,max_df=0.99) # хуже чем умолчательный\nvect = TfidfVectorizer()\nclf = SGDClassifier(loss='modified_huber',class_weight={0:1,1:10})\n#clf = SGDClassifier(loss='log')\n#svd = TruncatedSVD(n_components=100)\nmodel = Pipeline([('vect', vect),('clf', clf)])\npreds2 = cross_val_predict(model, df.question_text.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=-1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5eb0d9c06c6e52966c5dd464b2af55c64e64995b"},"cell_type":"code","source":"# подбираем границу\nfor i in np.arange(0.1,0.9,0.1):\n    print(\" {0} for bound {1}\".format(f1_score(df.target.values,preds2[:,1]>=i),i))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f0e3483bbf18fb886b78aea34452a6a2f67cd5ea"},"cell_type":"code","source":"print(classification_report(df.target.values,preds2[:,1]>=0.66)), roc_auc_score(df.target.values,preds2[:,1]), f1_score(df.target.values,preds2[:,1]>=0.66)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"127bf6875e2219bba36735dbba1d4b549ef5627a"},"cell_type":"markdown","source":"итак, they и why, которые планировалось убрать стоп-словами, выкинулись сами. What и how оказались сильнее"},{"metadata":{"trusted":true,"_uuid":"2aa589627777ae2416b7453f731ffa5bc5fc2239"},"cell_type":"code","source":"model.fit(df.question_text.values, df.target.values)\neli5.show_weights(model, top=10)\n#eli5.show_weights(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cb72c0bda0ac12f6905818068d191c4b8eb97664"},"cell_type":"code","source":"#vect = TfidfVectorizer(min_df=0.001,max_df=0.99,stop_words='english')\nvect = TfidfVectorizer(min_df=30,max_df=0.8,stop_words=stopwords.words('english'))\n#clf = SGDClassifier(loss='log')\n#svd = TruncatedSVD(n_components=100)\nmodel = Pipeline([('vect', vect),('clf', clf)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67d7498f3c7db17dd42d49567a9678c0c07e9b15"},"cell_type":"code","source":"%%time\npreds3 = cross_val_predict(model, df.question_text.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=4,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a5be07d0c510c0103796382d7ed9e146e503e4be"},"cell_type":"code","source":"# подбираем границу\nfor i in np.arange(0.1,1,0.1):\n    print(\" {0} for bound {1}\".format(f1_score(df.target.values,preds3[:,1]>=i),i))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e50cde4b6004f1a262a75e935a1d55faa53fbbe5","scrolled":true},"cell_type":"code","source":"print(classification_report(df.target.values,preds3[:,1]>=0.2)), roc_auc_score(df.target.values,preds3[:,1]), f1_score(df.target.values,preds3[:,1]>=0.2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a7eb97d2c0d3bda455ade61fe946a12a3251d21"},"cell_type":"code","source":"'5100' in vect.stop_words_, 'you' in vect.stop_words_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1301d86a11278e4c1d670c4268fce9028b66b244"},"cell_type":"code","source":"#import eli5\nmodel.fit(df.question_text.values, df.target.values)\n#eli5.show_weights(model, top=10)\neli5.show_weights(model)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48550e5ab18c18b7bb2d9430482b2fe017785b5d"},"cell_type":"code","source":"# old model without stop-words\nvect = TfidfVectorizer(min_df=0.001,max_df=0.99) \nmodel = Pipeline([('vect', vect),('clf', clf)])\nmodel.fit(df.question_text.values, df.target.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"63e6e6556850445e691a8a207d908f202b75d809"},"cell_type":"code","source":"eli5.show_prediction(clf, df.question_text[326527],vec=vect)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"4b38269e38ceabc23bb9ccb07c4483755f6d3307"},"cell_type":"code","source":"eli5.show_prediction(clf, df.question_text[22],vec=vect,targets=[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"894d079e359a1d59be0192eca98833ce2820df7d"},"cell_type":"code","source":"кляти bias","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"5a99e1e1e55abf39e12ad7a9ad34c64627e33162"},"cell_type":"code","source":"eli5.show_prediction(clf, df.question_text[45],vec=vect)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"458e03e45277f5459e7321c6a1639e4a6549762b"},"cell_type":"markdown","source":"# Часть 1.5 Стемминг и лемматизация"},{"metadata":{"_uuid":"376c8e8a6de73309d96de5d48f86095b134c2dd7"},"cell_type":"markdown","source":"Проведите для текстов стемминг и лемматизацию. Поскольку это процесс не быстрый, лучше сохранить предобработанные тексты в отдельный столбец DataFrame. Обучите модель на стеммированных и лемматизированных данных, сравните размер словаря"},{"metadata":{"trusted":true,"_uuid":"c9f57ef737d38f48a9bdb435bd8e229c4ef67a8a"},"cell_type":"code","source":"# эта вспомогательная функция поможет вам чуть быстрее пройтись по всему тексту\n# обратите внимание, что лемматизация будет в основном работать только для слов в нижнем регистре!\n\n@lru_cache(maxsize=2048)\ndef lemmatize_word(word):\n    parts = ['a','v','n','r']\n    lemmatizer = WordNetLemmatizer()\n    for part in parts:\n        temp = lemmatizer.lemmatize(word, part)\n        if temp != word:\n            return temp\n    return word    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58d3d552fb0ca1a99b1ead8991d5ee658bb08ba4"},"cell_type":"code","source":"stemmer = SnowballStemmer('english')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4051a5e5793c41252184244b463506fc4ac2e777"},"cell_type":"code","source":"def stem_q(question):\n    st=[]\n    for w in sklearn_tokenizer(question):\n        st.append(stemmer.stem(w))\n    return st","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"62256d60b7d9320fcb400f5d6309bdfefc6331db"},"cell_type":"code","source":"#vect = TfidfVectorizer(min_df=0.001,max_df=0.99)\nvect = TfidfVectorizer(stop_words='english')\nvect.fit_transform(df.question_text)\nsklearn_tokenizer = vect.build_tokenizer()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26d1fbcd9cac19bcd2dd90daed00cf86eaac8a85"},"cell_type":"code","source":"stem_q(df.question_text[11])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"255f1cc5e783d1cf05ed61cdaf9eecc7cc50eaf0"},"cell_type":"code","source":"stem_q(\"went goed appeared cats spring\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"52397d66076a7c29eb6f4c4149e226ed6eeaee2e"},"cell_type":"code","source":"df['stem'] = df.apply (lambda row: \" \".join(stem_q(row.question_text)),axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c220b469ed9d4d81155942753c9a171d8da0405f"},"cell_type":"code","source":"#df['stem']= df.apply (lambda row: \" \".join(row.stem), axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a491758c6598ad856558e070ff56351ecdeea26c"},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4cbd604ca9c87b1370e71df985db37ba548c8c9b"},"cell_type":"code","source":"%%time\npreds_st = cross_val_predict(model, df.stem.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d6a77e12eeed18649078127c50db785e2248ecd3"},"cell_type":"code","source":"print(classification_report(df.target.values,preds_st[:,1]>=0.3)), roc_auc_score(df.target.values,preds_st[:,1]), f1_score(df.target.values,preds_st[:,1]>=0.3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ba32d3647e17e74f7f18466d1125ee075fd15d0"},"cell_type":"code","source":"f1_score(df.target.values,preds2[:,1]>=0.3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"84dc7a540e05456b26963437d7d2e10240ee4be0"},"cell_type":"code","source":"model.fit(df.stem.values, df.target.values)\neli5.show_weights(model, top=15)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"afc29b43fa43b3929d1db26c2a6572bac055ae2e"},"cell_type":"code","source":"eli5.show_prediction(clf, df.stem[110],vec=vect)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1c9f182375d862f5ad0d34787587cdb311ab5504"},"cell_type":"markdown","source":"и снова кляти bias"},{"metadata":{"trusted":true,"_uuid":"44dfc8d49ccd0273b60232acf5c6dd9a5c74d13d"},"cell_type":"code","source":"len(vect.vocabulary_)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4cf27676539f4223f883382987cc1ad3fba3fc4d"},"cell_type":"code","source":"morestemmer = LancasterStemmer('english')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2bb7abedcbc81bcd9b8bef639c7ee6c2f097ea09"},"cell_type":"code","source":"def stem_ql(question):\n    st=[]\n    for w in sklearn_tokenizer(question):\n        st.append(morestemmer.stem(w))\n    return st","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"02bf5f08466f1a928d8affbe1b2fccacd5292f99"},"cell_type":"code","source":"print(stem_q(df.question_text[4078]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"085145d82db1cf35b311f9fb32fd58762fd86c85"},"cell_type":"code","source":"print(stem_ql(df.question_text[4078]))\nprint(stem_q(df.question_text[4078]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"21ab5837598ba63cfe054dc57d942754a9ea34cf"},"cell_type":"code","source":"def lemm(question):\n    le=[]\n    for w in sklearn_tokenizer(question):\n        le.append(lemmatize_word(w.lower()))\n    return le","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b5775c78e44471c4ade8c6fd0265793cc1edba00"},"cell_type":"code","source":"lemm(df.question_text[56])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b6acc93189e8f763aaaeb2ff726e6bc97f4168da"},"cell_type":"code","source":"def stemlem_q(question):\n    st=[]\n    for w in sklearn_tokenizer(question):\n        st.append(stemmer.stem(lemmatize_word(w.lower())))\n    return st","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"73b24df1cb4e750337ef9bbeaf0d8bd092dc3f09"},"cell_type":"code","source":"stemlem_q(df.question_text[46])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3afbc4a324440c9a799e1696835d95268fa410df"},"cell_type":"code","source":"df['lemmstem'] = df.apply (lambda row: \" \".join(stemlem_q(row.question_text)),axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"444cec366e9bdef84635603ea5d365977e90fbe8"},"cell_type":"code","source":"df.to_csv(\"df_to_csv.csv\",index=False,sep=',')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45a883b68c4c2ea3e7fb7fb4cb0a41aeb4f0256d"},"cell_type":"code","source":"import pickle\nwith open('dfwithlem.pkl','wb') as f: pickle.dump(df,f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"86e8e927d2304bb4b4a1d1e22ff43cdbf0e68aa8"},"cell_type":"code","source":"from IPython.display import HTML\nimport base64\n\ndef create_download_link(data, title, filename):\n    csv = data.to_csv()\n    b64 = base64.b64encode(csv.encode())\n    payload = b64.decode()\n    html = f'<a target=\"_blank\">{title}</a>'\n    return HTML(html)\n\ncreate_download_link(df, \"darou\", \"data.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"11838a36857f0f55dae9521eeaa156adb1f6c292"},"cell_type":"code","source":"from IPython.display import HTML\nimport pandas as pd\nimport numpy as np\nimport base64\n\n# function that takes in a dataframe and creates a text link to  \n# download it (will only work for files < 2MB or so)\ndef create_download_link(df, title = \"Download CSV file\", filename = \"data.csv\"):  \n    csv = df.to_csv()\n    b64 = base64.b64encode(csv.encode())\n    payload = b64.decode()\n    html = '<a download=\"{filename}\" href=\"data:text/csv;base64,{payload}\" target=\"_blank\">{title}</a>'\n    html = html.format(payload=payload,title=title,filename=filename)\n    return HTML(html)\n\n# create a link to download the dataframe\ncreate_download_link(df,'ui','data.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e12602120ad8bb2874dc288db25c3f7bfbe54d0"},"cell_type":"code","source":"df.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9f7881fcdb2bef97538dd1de58ba1208a0707a3c"},"cell_type":"code","source":"vect = TfidfVectorizer(min_df = 0.001, max_df=0.99)\nclf = SGDClassifier(loss='modified_huber',class_weight={0:1,1:10})\nmodel = Pipeline([('vect', vect),('clf', clf)])\npreds_st = cross_val_predict(model, df.lemmstem.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5fbd2ed941ccd8e36efd0f6c41ddaeedefd25100"},"cell_type":"code","source":"# подбираем границу\nfor i in np.arange(0.1,0.9,0.1):\n    print(\" {0} for bound {1}\".format(f1_score(df.target.values,preds_st[:,1]>=i),i))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f7c06bbe932234b5cf160ebe56859bbeee0e0efb"},"cell_type":"code","source":"vect = TfidfVectorizer(min_df=0.001,max_df=0.99)\nclf = SGDClassifier(loss='log')\n#svd = TruncatedSVD(n_components=100)\nmodel = Pipeline([('vect', vect),('clf', clf)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"edb36f048e38cb122699c5bba5ddfd38cfe81663"},"cell_type":"code","source":"%%time\npreds_lm = cross_val_predict(model, df.lemm.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e8176c885b633ec0da778641ee73ddd9b4a7cc8e"},"cell_type":"code","source":"print(classification_report(df.target.values,preds_lm[:,1]>=0.3)), roc_auc_score(df.target.values,preds_lm[:,1]), f1_score(df.target.values,preds_lm[:,1]>=0.3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eda5edf233f14dfcd4cbbc1e913375bf9e356f3d"},"cell_type":"code","source":"model.fit(df.stem.values, df.target.values)\neli5.show_weights(model, top=15)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c72eb54c4b00c31b3c5c7f31bb3906c1dded8a9b"},"cell_type":"code","source":"print(lemmatize_word('having'))\nprint(stemmer.stem('having'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"3de87867ad7919cb4cc9a1ef104dea9efce02765"},"cell_type":"code","source":"print(stemmer.stem('tolerant'))\nprint(stemmer.stem('tolerable'))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6bef009d7e9ab39a44711f77c459908fe5243777"},"cell_type":"markdown","source":"# Часть 1.5 Творчество"},{"metadata":{"_uuid":"54b610d4e91ce1dbdcceedce66f52fe84d714c5c"},"cell_type":"markdown","source":"Подберите с помощью GridSearch или RandomizedSearch хорошие параметры вашей модели. Когда вы делаете сетку для пайплайна параметры надо называть так как у внутреннего класса, добавляя названия соответствующего этапа перед ними. Например:"},{"metadata":{"trusted":true,"_uuid":"93798b15991fcec82ae8703987fcf7c7000070c5"},"cell_type":"code","source":"grid = {'clf__l1_ratio': [0.1, 0.5],\n        'clf__loss': ['log', 'modified_huber'],\n        'vect__min_df': [0.01, 0.03,0.005],\n        'vect__norm':['l1','l2'],\n        'vect__stop_words':[stops,'english',None]\n        }","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"451175e88d1b75cd6c5599feffbafd968f512c67"},"cell_type":"code","source":"pipe = Pipeline([('vect',TfidfVectorizer(max_df=0.99)),('clf', SGDClassifier(class_weight={0:1,1:15}))])\nsearch = RandomizedSearchCV(pipe,param_distributions=grid, n_iter = 20, return_train_score=True,\n                      cv=StratifiedKFold(4), verbose=1,\n                      scoring='f1', n_jobs=-1)\n#search = RandomizedSearchCV(pipe,param_distributions=grid, n_iter = 20,\n#                      cv=StratifiedKFold(4), verbose=1,\n #                     scoring='f1', n_jobs=-1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb4d4973fe2fbc32d8aefc2729cc065b20b62527"},"cell_type":"code","source":"search.fit(df.stem.values,df.target.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"37369bec196ece06bfddf9af44adfde09143b92f"},"cell_type":"code","source":"theModel=search.best_estimator_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2664ee901259f7edc02e2485d7655cda3897bfcf"},"cell_type":"code","source":"theModel","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4b23f45c5430e66d6a94cb3fb3363f736a77aac5"},"cell_type":"code","source":"search.best_params_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d462ca602bb82322d4a2430cefc34d97c11cf55"},"cell_type":"code","source":"{'vect__stop_words': None,\n 'vect__norm': 'l2',\n 'vect__min_df': 0.005,\n 'clf__loss': 'modified_huber',\n 'clf__l1_ratio': 0.1}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2a2caa05ed24675e2febc906aec1e13d6d194d42"},"cell_type":"code","source":"search.best_score_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"24a5ef5ce35319ed6215d9a9a68168b9279c6a17"},"cell_type":"code","source":"preds = cross_val_predict(theModel, df.stem.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e99dae90e7b2bab338ecb9df10ba84fc9ed1d34f"},"cell_type":"code","source":"print(classification_report(df.target.values,preds[:,1]>=0.5)), roc_auc_score(df.target.values,preds[:,1]), f1_score(df.target.values,preds[:,1]>=0.5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8bbd8e113d5adabb383de7867596175c4e1e6c27"},"cell_type":"code","source":"f1_score(df.target.values,preds[:,1]>=0.8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6fbccae504c37bcbecae9b7f95a0e194da020086"},"cell_type":"code","source":"theModel.fit(df.stem.values, df.target.values)\neli5.show_weights(theModel, top=15)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1b2231d5c07a016df26b60bb9057df0ab3621b7"},"cell_type":"code","source":"eli5.show_prediction(theModel, df.stem[6],vec=vect,targets=[1])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bf5989e36d13d24215f5553dc33de1eb558be48c"},"cell_type":"markdown","source":"Попробуйте разные функции ошибок для SGD. Например hinge (внутри получится svm, нельзя будет предсказывать вероятности), modified_huber - комбинация l1 и l2 ошибки для классификации (упоминалась на лекции). \nРазную предобработку, параметры, подберите порог принятия решения.\n\nПопробуйте уменьшать размер словаря путем комбинации лемматизации (или стемминга) и увеличения min_df. Проанализируйте, как меняется размер словаря, а также качество модели."},{"metadata":{"trusted":true,"_uuid":"c4ac2ba655e3602fdf5aca90759a9e023d25bd1b"},"cell_type":"code","source":"stops=stopwords.words('english')\n# вот эти слова таки выглядят довольно полезными\nstops.remove('why')\nstops.remove('what')\nstops.remove('how')\nstops","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d89735764c66e193f102d5bfe14faced2b8338d"},"cell_type":"code","source":"from sklearn.feature_extraction.text import HashingVectorizer\n\nhvec = HashingVectorizer(stop_words=stops, ngram_range=(1,2))\nclf = SGDClassifier(max_iter=10, random_state=42, loss='log')\npipe = Pipeline([('hvec',hvec),('clf', clf)])\n#pipe.fit(df.question_text, df.target)\npreds = cross_val_predict(pipe, df.question_text.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2488143071574d20c70f1bd86f3405dedef47c76"},"cell_type":"code","source":"print(classification_report(df.target.values,preds[:,1]>=0.3)), roc_auc_score(df.target.values,preds[:,1]), f1_score(df.target.values,preds[:,1]>=0.3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"14b934fefa1022ee52dd292af0fca5b7947fb3aa"},"cell_type":"code","source":"vect = TfidfVectorizer(min_df=0.001,max_df=0.99,stop_words = stops) \nclf = SGDClassifier(loss='log')\n#svd = TruncatedSVD(n_components=100)\nmodel = Pipeline([('vect', vect),('clf', clf)])\npreds = cross_val_predict(model, df.question_text.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"072121191c52943049326856b97e4ea662f0349b"},"cell_type":"code","source":"print(classification_report(df.target.values,preds[:,1]>=0.3)), roc_auc_score(df.target.values,preds[:,1]), f1_score(df.target.values,preds[:,1]>=0.3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f921fd57b2caaadc5fbb9ac60ae8fb44d9d65ec"},"cell_type":"code","source":"len(df.target[df.target==0])/len(df.target[df.target==1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"3c6fdda0e86ec3e8a0621d946dfd40a794399a62"},"cell_type":"code","source":"vect = TfidfVectorizer(min_df=0.001,max_df=0.99,stop_words=stops)\n#vect = TfidfVectorizer(min_df=0.001,max_df=0.99)\nclf = SGDClassifier(loss='log', class_weight={0:1,1:15})\n#svd = TruncatedSVD(n_components=100)\nmodel = Pipeline([('vect', vect),('clf', clf)])\npreds = cross_val_predict(model, df.question_text.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"86a35c8bfa383ec490084282e8e357988088c0ec"},"cell_type":"code","source":"print(classification_report(df.target.values,preds[:,1]>=0.8)), roc_auc_score(df.target.values,preds[:,1]), f1_score(df.target.values,preds[:,1]>=0.8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7e26d7d99c05ab5064840d58730b6831c88d6729"},"cell_type":"code","source":"f1_score(df.target.values,preds[:,1]>=0.8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"15eb862c9bdf109be1ba98a11b323524251b1a04"},"cell_type":"code","source":"preds[1:20]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c653450cf8665072918466c71fe77ac8e46f3d91"},"cell_type":"code","source":"preds[6]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa780fe5154d2f4267de1a2e92c66f7a99383027"},"cell_type":"code","source":"df[6:7]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"64b5d73590462699c208107c202510452e184278"},"cell_type":"code","source":"eli5.show_prediction(clf, df.question_text[6],vec=vect,targets=[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"4793821a679664dd9f31d317861cdc5d5c838cb7"},"cell_type":"code","source":"model.fit(df.question_text.values, df.target.values)\neli5.show_weights(model, top=15)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"54dc627348cb68492ffd4768f2689d19c1ada1c6"},"cell_type":"code","source":"vect = TfidfVectorizer(min_df=0.001,max_df=0.99,stop_words=stops)\n#vect = TfidfVectorizer(min_df=0.001,max_df=0.99)\nclf = SGDClassifier(loss='log', class_weight={0:1,1:15})\n#svd = TruncatedSVD(n_components=100)\nmodel = Pipeline([('vect', vect),('clf', clf)])\npreds_st = cross_val_predict(model, df.stem.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1e1976aabf0ffd8628b710621378472e78dd55bf"},"cell_type":"code","source":"print(classification_report(df.target.values,preds_st[:,1]>=0.8)), roc_auc_score(df.target.values,preds_st[:,1]), f1_score(df.target.values,preds_st[:,1]>=0.8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ef106c86202d86259ab0776e958cb8078e177f8c"},"cell_type":"code","source":"model.fit(df.stem.values, df.target.values)\neli5.show_weights(model, top=40)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f5fa4c6b2269b3515ba59509e96e2410a2b4230e"},"cell_type":"code","source":"vect = TfidfVectorizer(min_df=0.001,max_df=0.99,stop_words=stops)\n#vect = TfidfVectorizer(min_df=0.001,max_df=0.99)\nclf = SGDClassifier(loss='log', class_weight={0:1,1:15})\n#svd = TruncatedSVD(n_components=100)\nmodel = Pipeline([('vect', vect),('clf', clf)])\npreds_st = cross_val_predict(model, df.stem.values, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"415f90e8fc92e940d37548e44833a37e10ee414c"},"cell_type":"markdown","source":"От количества творчества и проведенных экспериментов будет зависеть ваша оценка"},{"metadata":{"_uuid":"1e0c30df7313384d64a6589f356c984c1192cba1"},"cell_type":"markdown","source":"# Часть 2 (На 10). Тематическое моделирование"},{"metadata":{"_uuid":"8a307c931373c6e41a26d51889d6562a9c39ba5b"},"cell_type":"markdown","source":"Используйте либо https://scikit-learn.org/stable/modules/generated/sklearn.decomposition.LatentDirichletAllocation.html, либо модели из gensim для тематического моделирования на этих текстах. Посмотрите на полученные темы, попытайтесь их интерпретировать. Используйте распределение тем для документа в качестве набора признаков, описывающих его. Обучите модель на этих признаках и сравните с tf-idf.\n\nПоскольку задание на 10 тут не будет никакой дополнительной информацию. Те, кто делают, пусть пишут мне письма и задают вопросы на паре. Отвечу и помогу точечно."},{"metadata":{"trusted":true,"_uuid":"f36a5732cf4ac679e821745a4627d7767c6306d3"},"cell_type":"code","source":"from sklearn.decomposition import LatentDirichletAllocation","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a7ba8de30a1df1b878c3fba4ba95ed03908bf582"},"cell_type":"markdown","source":"https://towardsdatascience.com/topic-modeling-and-latent-dirichlet-allocation-in-python-9bf156893c24 "},{"metadata":{"_uuid":"991b4e1535fffe38188f16e52990bf74061389b6"},"cell_type":"markdown","source":"1. Токенизация"},{"metadata":{"trusted":true,"_uuid":"5f2824cf66fb2ae4d65dccd00e4610129d1fdf90"},"cell_type":"code","source":"vect = TfidfVectorizer(min_df=0.001,max_df=0.99,stop_words=stopwords.words('english'))\nvect.fit(df.question_text)\nsklearn_analyzer = vect.build_analyzer()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba57b5752d457fba975df48ef790f124228ae697"},"cell_type":"code","source":"from sklearn.datasets import make_multilabel_classification\nX, _ = make_multilabel_classification(random_state=0)\nX","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c25ca58f3efa8f36c7d165bf51eeb66b3ba9fd94"},"cell_type":"code","source":"X.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eb9c0213663034bdef69a75e33ff9590c363e20b"},"cell_type":"code","source":"lda = LatentDirichletAllocation(n_components=5, learning_method = 'online', random_state=0)\nlda.fit(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7bd7931fa3917885381ee5a53563014018ae39c2"},"cell_type":"code","source":"lda.components_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d5ae29d42b398b22e415c273e613268f449ae67c"},"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c0afc03e7debf6e773f2fe3eb7e32ed62f9fa6f"},"cell_type":"code","source":"tf_vectorizer = CountVectorizer(max_df=0.99, min_df=0.001,\n                                max_features=1000,\n                                stop_words='english')\ntf = tf_vectorizer.fit_transform(df.question_text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"284cd89e7a42cfb77488850fa4211160c2331580"},"cell_type":"code","source":"analyzer = tf_vectorizer.build_analyzer()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d3eb257281b04159cf967018e8c11f6967e4eb1"},"cell_type":"code","source":"len(df_small.question_text)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5cf0e98ef6082796eeb7e0e48f3b18d996a52771"},"cell_type":"code","source":"tf.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae2328a60dfb454b7bf606d7bf6559a80f60b296"},"cell_type":"code","source":"lda = LatentDirichletAllocation(n_components=2, max_iter=5,\n                                learning_method='online',\n                                learning_offset=10.,\n                                random_state=0)\nlda.fit(tf)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5101d78d68caf97c59cdb1041fa7aa32daab7ba"},"cell_type":"code","source":"def print_top_words(model, feature_names, n_top_words):\n    for topic_idx, topic in enumerate(model.components_):\n        message = \"Topic #%d: \" % topic_idx\n        message += \" \".join([feature_names[i]\n                             for i in topic.argsort()[:-n_top_words - 1:-1]])\n        print(message)\n    print()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5bb31ae8540618bf1390b7faef8d6067e7b2fc4b"},"cell_type":"code","source":"tf_feature_names = tf_vectorizer.get_feature_names()\nprint_top_words(lda, tf_feature_names, 20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42e86b5f44f893ed422b35e265d40a20cd7bd28e"},"cell_type":"code","source":"tf_vectorizer.transform([df.question_text[3]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"11865886854594ebf794d5c5dbeea98f944b6fc1"},"cell_type":"code","source":"lda.transform(tf_vectorizer.transform([df.question_text[3]]))[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9ef781920106050dde8fad73e289fd2d763038d2"},"cell_type":"code","source":"df[3:4]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f2bb068dd6309d30d9f2e00f252052bedf4e14ae"},"cell_type":"code","source":"lda.transform(tf_vectorizer.transform(df_small.question_text[671:700]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44ff4779bbca436a5a7129941970466a3f72229c"},"cell_type":"code","source":"df_small[671:700]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c03207f3d4cb34a8e66386aca3b6f630963167ac"},"cell_type":"code","source":"def get_pred(question):\n    ar = lda.transform(tf_vectorizer.transform([question]))[0]\n    num = ar[np.argpartition(ar,-2)[-2:]].sum()\n    if num>0.75:\n        return 1\n    else:\n        return 0\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bcfdef0156981b07f71f93cbc1ba381c86b503d8"},"cell_type":"code","source":"get_pred(df.question_text[3])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58b2c5a21cfa2c7fdf90e5ec041019120932430f"},"cell_type":"code","source":"len(df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f2c39671e286f24e5e1815a99e9644147a5a4890"},"cell_type":"code","source":"%%time\n\npreds = list(map(get_pred, df.question_text[:100000]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"182cc99991ce408621e15bc77fb429b700cad0f8"},"cell_type":"code","source":"f1_score(preds,df.target[:100000])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae0fbe7585a8ac4543bb1f4e9e1cb35c2961439e"},"cell_type":"code","source":"np.array([8,9,7,8])[np.argpartition(np.array([8,6,0,9,-8,4]),-2)[-2:]].sum()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7b03bb09b4993cee87091b392c450bddfa52e468"},"cell_type":"code","source":"tfidf_vectorizer = TfidfVectorizer(max_df=0.99, min_df=2,\n                                   max_features=n_features,\n                                   stop_words='english')\nt0 = time()\ntfidf = tfidf_vectorizer.fit_transform(data_samples)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"475801b28b5eb13820534094d283c5319b4a639d"},"cell_type":"markdown","source":"2. Лемматизация + стемминг"},{"metadata":{"trusted":true,"_uuid":"0f4a333e10d4f0fefaed2f8e23caa879d18484cd"},"cell_type":"code","source":"def stemlem_q(question):\n    st=[]\n    for w in sklearn_tokenizer(question):\n        st.append(stemmer.stem(lemmatize_word(w.lower())))\n    return st","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b58739a9ac5ad269465c3b7f2e561a48a9343328"},"cell_type":"code","source":"stemlem_q(df.question_text[7])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"536d1c3f669924f27d41c6d5b8fba9d0fb310173"},"cell_type":"code","source":"df['stemlem'] = df.apply (lambda row: \" \".join(stemlem_q(row.question_text)),axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3050ff383e73dc57f750e8d4545ae05da2d336e"},"cell_type":"code","source":"another","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a55e4ab426124468632ced8ec464050a1730b5f"},"cell_type":"code","source":"import re\nimport spacy\nimport gensim\nimport gensim.corpora as corpora\nfrom gensim.utils import simple_preprocess\nfrom gensim.models import CoherenceModel","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e560d98f6171050971fd91c0912664673ff0e345"},"cell_type":"code","source":"stop_words = stopwords.words('english')\n\ndef lemm1(question):\n    le=[]\n    for w in sklearn_tokenizer(question):\n        if w not in stop_words: le.append(lemmatize_word(w.lower()))\n    return le","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"02552d2f04f5276da1b2c3bae526d8fc5a0ea87b"},"cell_type":"code","source":"df['lemm'] = df.apply (lambda row: \" \".join(lemm1(row.question_text)),axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d16a23a927c07942c3a259cf569ccb47a3e8a1f"},"cell_type":"code","source":"vect.fit_transform(df.lemm)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d1e4c52cf79d44d0ab67cc64fe99c1f6fb0d06e4"},"cell_type":"code","source":"vect.vocabulary_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9519d3436e1b8873d35d74f0e539b59724493198"},"cell_type":"code","source":"X=vect.fit_transform(df.lemm)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f4fe627b5f5fa3d89f4050683d0d8c005c8521a"},"cell_type":"code","source":"print(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"214c6d4165f6370c944519f322a369b425e3dfc0"},"cell_type":"code","source":"id2word = corpora.Dictionary(sklearn_tokenizer(text) for text in df.lemm)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e60980f5ea10469d39d23da03d0f5cadf5ccc6d5"},"cell_type":"code","source":"corpus = [id2word.doc2bow(sklearn_tokenizer(text)) for text in df.lemm]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"057f4fc5119402fde395cf7862e9b35fe6a8b975"},"cell_type":"code","source":"print(corpus[:1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a458a1105906eee67bc047a9897b66afa1789ba"},"cell_type":"code","source":"lda_model = gensim.models.ldamodel.LdaModel(corpus=corpus,\n                                           id2word=id2word,\n                                           num_topics=20, \n                                           random_state=100,\n                                           update_every=1,\n                                           chunksize=100,\n                                           passes=10,\n                                           alpha='auto',\n                                           per_word_topics=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"95964a64315fde1772212fc924a1603ba044e89d"},"cell_type":"markdown","source":"# Часть 3. word2vec"},{"metadata":{"_uuid":"656289ad9322d0a3085a4e89d4f4b387bb79615d"},"cell_type":"markdown","source":" #  Word2vec # "},{"metadata":{"trusted":true,"_uuid":"e2aca4fb28863ca00bc9c18fd1c0466fc08838d2"},"cell_type":"code","source":"from gensim.models import KeyedVectors\nfrom gensim.models import Doc2Vec\nfrom gensim.models.doc2vec import TaggedLineDocument, TaggedDocument","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ae46d91925d3f964b17ab07a08c7128638ee013"},"cell_type":"code","source":"w2v = KeyedVectors.load_word2vec_format('../input/vec2wordslim/GoogleNews-vectors-negative300-SLIM.bin', binary=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f894bb5ba6245f0abf9b75223d7bf851a8cd8e9d"},"cell_type":"code","source":"data = [\"I love machine learning. Its awesome.\",\n        \"I love coding in python\",\n        \"I love building chatbots\",\n        \"they chat amagingly well\"]\n\ntagged_data = [TaggedDocument(words=word_tokenize(_d.lower()), tags=[str(i)]) for i, _d in enumerate(data)]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d2a4d3166d3f29a9d7469e9479d17f520a35aa38"},"cell_type":"code","source":"vect(\"I love chatbots\".lower())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"85ab6aecc541434b9a039cb4a3bdbedf325b107a"},"cell_type":"code","source":"vect = TfidfVectorizer(min_df=0.001,max_df=0.99) \nX=vect.fit_transform(df.question_text)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"26f7d1da5e6491631099af5039f1379af3c17429"},"cell_type":"code","source":"vect.get_feature_names()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"39eab42e35c62555d9c929c423f9ecdeacef082a"},"cell_type":"code","source":"t_data=[TaggedDocument()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25d63ed220698df47db4b793ea77216f3e2cd5c9"},"cell_type":"code","source":"t_data=[TaggedDocument(words=word_tokenize(_d.lower()), tags=[str(i)]) for i, _d in enumerate(df.question_text)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a128d5b7edfc1dcaede8440092b46223e7a4f166"},"cell_type":"code","source":"tagged_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7abae2fc819b1dd59ba8e5e40b175a05dd3fcb1d"},"cell_type":"code","source":"#model = Doc2Vec(size=vec_size,alpha=alpha, min_alpha=0.00025,min_count=1,dm =1)\n\ndocvec = Doc2Vec(tagged_data, vector_size=5, window=2, min_count=1, workers=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9816f9af2aed2138702780ba8ee1ab7c01b42db2"},"cell_type":"code","source":"docvec.docvecs.most_similar('1')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c785e000d8217cb626e45dbe9e6eb27265075eff"},"cell_type":"code","source":"from sklearn.svm import OneClassSVM","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5370fb972671120924bd7f0e2d09e1ecb8eddeca"},"cell_type":"code","source":"svm=OneClassSVM()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e01831841b938abe52861b6858544dc8cb096873"},"cell_type":"markdown","source":"В этой части вам надо будет пощупать вектора слов, поискать ассоциации, а потом использовать эти вектора как признаки для вашей модели.\n"},{"metadata":{"_uuid":"b088ab1b424a191b12052129bf4818d164a89018"},"cell_type":"markdown","source":"Скачайте векторы отсюда и загрузите их при помощи кода ниже.\n\nhttps://drive.google.com/uc?id=0B7XkCwpI5KDYNlNUTTlSS21pQmM&export=download\n\n\nЕсли ваш компьютер не позволяет загрузить большие векторы, воспользуйтесь векторами вот отсюда. Они будут в 10 раз меньше по размеру.\n\nhttps://github.com/eyaler/word2vec-slim/blob/master/GoogleNews-vectors-negative300-SLIM.bin.gz\n\nДля работы с большими векторами нужно как минимум 8 гб оперативной памяти. Если у вас их нет, то пользуйтесь маленькими, либо google colaboratory, либо kaggle kernels."},{"metadata":{"trusted":true,"_uuid":"725eddbbe6ecc703aa926e1fcff176cee6be9207"},"cell_type":"code","source":"from gensim.models import KeyedVectors","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"ccfdd7ee37cec071ac57bd204d710d43ca21da8d"},"cell_type":"code","source":"w2v = KeyedVectors.load_word2vec_format('LB04/GoogleNews-vectors-negative300-SLIM.bin', binary=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"09b545cf647755d637521e22ee52f342a66d24f7"},"cell_type":"markdown","source":"# Задание 3.1 Знакомство с векторами"},{"metadata":{"_uuid":"48bdcd1203a249327922fb9b1deb52611b34e242"},"cell_type":"markdown","source":"Пощупайте векторы и найдите по крайней мере 2 **семантических** и 2 **синтаксических** аналогии"},{"metadata":{"trusted":true,"_uuid":"a8e4dfdbd930935e3c7a7d9add4f664eb0e281dd"},"cell_type":"code","source":"# вот так можно получить вектор слова\nw2v['Minsk']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0e40a55880b5489d12ba5b01e79b822778089a3e"},"cell_type":"markdown","source":"Вот так работают аналогии"},{"metadata":{"trusted":true,"_uuid":"357ca821d2fefd8129c5ae1ec72f76226ff90ead"},"cell_type":"code","source":"# vec(Minsk) - vec(Belarus) + vec(Russia) ~= vec(Moscow)\nw2v.most_similar(positive=['Minsk', 'Russia'], negative=['Belarus'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d8e1ebd8f0a5074cec83d65d4c75b6f78ae2dba9"},"cell_type":"code","source":"w2v.most_similar(positive=['earth', 'blue'], negative=['sky'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0bcd353ca49451354aa2e155d54df7f5712055bc"},"cell_type":"code","source":"w2v.most_similar(positive=['english', 'China'], negative=['America'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"392e01b0d2292bb7891641355ae99a78ae50a2a5"},"cell_type":"code","source":"w2v.most_similar(positive=['grow', 'wondered'], negative=['wonder'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cd19a3d6b387075a5d4f9245d19c97955a8de36f"},"cell_type":"code","source":"w2v.most_similar(positive=['worse', 'large'], negative=['bad']) ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"67eabbf85b2227b9143ae1db51d9dc8ec6b0877e"},"cell_type":"markdown","source":"# Задание 3.2 Использование среднего вектора в качестве фичей для моделирования"},{"metadata":{"_uuid":"5132dbe93564cf96015633c25a58834a82c49ef3"},"cell_type":"markdown","source":"Как мы и говорили на лекции, вектора слов можно использовать для построения признакового описания текста путем усреднения векторов. Сделайте это. Для этого нужно:\n1. Токенизировать каждый текст\n2. Для каждого токена попробовать найти вектор в w2v. Если его там нет, то просто игнорировать.\n3. Усреднить все вектора слов текста для получения его признакового описания. Если ни одного из слов в тексте нет вектора, то вернуть вектор из нулей той же длины\n\nНа полученных признаках обучите модель для предсказания и сравните качество.\n\nВНИМАНИЕ! Не забывайте нормировать вектора!\n"},{"metadata":{"trusted":true,"_uuid":"57081af313b221bd9969fc9c167ceae1eceab223"},"cell_type":"code","source":"np.linalg.norm(w2v.wv.word_vec('cats',use_norm=True))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51950041ef80bb5bf374703173753135d58b0f33"},"cell_type":"code","source":"w2v.wv.word_vec('cats',use_norm=True)-w2v.wv.word_vec('cat',use_norm=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"41c227545c9efd6251b8efe148e7609fd2f67571"},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ea027b4e466400796deeb8a46174805ac0c2fb65"},"cell_type":"code","source":"vect = TfidfVectorizer(min_df=0.001,max_df=0.99)\nvect.fit(df.lemmstem.values)\nsklearn_tokenizer = vect.build_tokenizer()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9936dc975194a295226bd4685f6be7454ca488ec"},"cell_type":"code","source":"def w2vquestion(question):\n    ar=[]\n    for w in sklearn_tokenizer(question):\n        if w in w2v: ar.append(w2v.wv.word_vec(w,use_norm=True))\n    \n    if len(ar) == 0:\n        ar = np.zeros(300)\n    else:\n        ar = np.array(ar).mean(axis=0)\n    return ar","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"11fcfc7f6242f17c5a3e7a1ccb8d1d3d6fe7b513"},"cell_type":"code","source":"w2vquestion(df.lemmstem[5])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e49d89fc4928d5cc041e430fdad4d242ac90b24"},"cell_type":"code","source":"%%time\n\nw2vquestion(df[\"question_text\"].values[24])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4cfd6e7c738b33c45c59ec53a2ff4be6fb9a596d"},"cell_type":"code","source":"train_x = list(map(w2vquestion, df.lemmstem.values))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1df59060d73fd4ff7a6f4aaa7f435803fb69f459"},"cell_type":"code","source":"np.shape(train_x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0bff591b887ae8c9816744bd14c2a7bf2e50005b"},"cell_type":"code","source":"#clf = SGDClassifier(loss='epsilon_insensitive', class_weight={0:1,1:15},epsilon = 0.01)\n# ну такое, 0.4 и хуже\nclf = SGDClassifier(loss='modified_huber', class_weight={0:1,1:15}) #збс пока\n#clf.fit(train_x,df.target[:1000000])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a8998bd7bdf237e0f77366f625fa0d66c43398f"},"cell_type":"code","source":"preds = cross_val_predict(clf, train_x, df.target.values,\n                          cv=StratifiedKFold(4), n_jobs=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ebdb92f29788067a5d91fc733fd802c2aa5e6ff"},"cell_type":"code","source":"preds[1:9]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e810ec57b788683ddb0480deb4240b1bee946c84"},"cell_type":"code","source":"test_x = list(map(w2vquestion, df[\"question_text\"].values[1000000:]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"19061aa5aa7a054f22c1d8d3d26a9768cbe4ec88"},"cell_type":"code","source":"f1_score(preds[:,1]>=0.78,df.target.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a951624d0c60ab4ffb14030c06e0ef8bff3c72f3"},"cell_type":"code","source":"# подбираем границу\nfor i in np.arange(0.1,0.9,0.1):\n    print(\" {0} for bound {1}\".format(f1_score(df.target.values,preds[:,1]>=i),i))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f365e4d660fcdc50a6f61d73ccdc5a0a2a036794"},"cell_type":"markdown","source":"# Задание 3.3 (На 10) Обучение doc2vec"},{"metadata":{"_uuid":"1a3b27f50eb7752cf641af690f4f25fd6bb54091"},"cell_type":"markdown","source":"В этом опциональном задании вам нужно обучить модель для получения векторов текстов, поискать похожие тексты, а также использовать вектор текста в качестве признаков для предсказания в этой же задаче. Тот кто берется за это задание, высылает мне запрос, а я присылаю пример, как это делать.\n"},{"metadata":{"trusted":false,"_uuid":"b7802481da9a6e1e88505989cc7e18b4fd2204e7"},"cell_type":"code","source":"from gensim.models import Doc2Vec\nfrom gensim.models.doc2vec import TaggedLineDocument","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5db0f1e48bbe91368d098e9e68803e243236cf63"},"cell_type":"markdown","source":"# Часть 4. Участие в конкурсе на kaggle"},{"metadata":{"_uuid":"c087d304f3071e159eabec7d42af30fbaf672db8"},"cell_type":"markdown","source":"Используйте ваши лучшие наработки локально, чтобы забраться как можно выше на доске лидеров в соревновании. Для этого нужно перенести часть кода в kaggle kernels. Для этого нужно разобраться с https://www.kaggle.com/c/quora-insincere-questions-classification#Kernels-FAQ"},{"metadata":{"_uuid":"dbbff8213ebc529972561eac6ed9a88510b82411"},"cell_type":"markdown","source":"Возможно, ваша лучшая модель будет какой-то из атомарных моделей выше. А, возможно, у вас получится удачно смешать модели на разных признаках. Итоговое место и оценка зависит только от вас"},{"metadata":{"trusted":true,"_uuid":"14f672b016d74c3a305135528a66aa3e9cbadef0"},"cell_type":"code","source":"svm.fit(X,df.target.values)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}