{"cells":[{"metadata":{"_uuid":"a163a68990705b191307d34ba06ff16503538526"},"cell_type":"markdown","source":"## 1.1 Загрузка необходимых библиотек"},{"metadata":{"trusted":true,"_uuid":"57d9096bada4857e9b7b97b77475978e4fd37a23"},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\n\nfrom nltk.corpus import stopwords \nfrom nltk.tokenize import word_tokenize \n\nimport nltk\nnltk.download('stopwords')\nnltk.download('punkt')\n\nfrom collections import Counter\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.metrics import f1_score\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import GridSearchCV\nfrom collections import OrderedDict\nfrom sklearn.metrics import roc_auc_score, log_loss\nimport itertools","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"254b32eecd5f8316a128193e8cb28cd51e708764"},"cell_type":"code","source":"data_train_gen = pd.read_csv(\"../input/train.csv\")\ndata_test = pd.read_csv(\"../input/test.csv\")\nprint(\"Train datasets shape:\", data_train_gen.shape)\nprint(\"Test datasets shape:\", data_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"24c552b9b001f834a55181f3084962e9b7c8d0e9"},"cell_type":"code","source":"data_train_gen.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"788a327f7e71e3382c74332fe1a4b23915fd8eaa"},"cell_type":"markdown","source":"##  1.2 Посмотрим на число вхождений"},{"metadata":{"trusted":true,"_uuid":"2506525a76ee56b32b649f7abb67369b61c7b1c9"},"cell_type":"code","source":"NUMBER_OF_1_LABELS = data_train_gen.target.value_counts()[1]\nNUMBER_OF_1_LABELS","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e36280f50f4f6e1de76339bcbc36910a52cd18fc"},"cell_type":"code","source":"count_Class=pd.value_counts(data_train_gen.target, sort= True)\ncount_Class.plot(kind= 'bar', color= [\"blue\"])\nplt.title('Bar chart')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6db3dd0955cd0c351ca60e7ff5ef839784f62d9e"},"cell_type":"code","source":"count_Class.plot(kind = 'pie',  autopct='%1.0f%%')\nplt.title('Pie chart')\nplt.ylabel('')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ae15e0c4c145b24ef0557be9988793fac693d363"},"cell_type":"markdown","source":"## 1.3 Сбалансируем выборку"},{"metadata":{"trusted":true,"_uuid":"0dc0220ba38e8558f2d9dd88f193e14504adf2d4"},"cell_type":"code","source":"df1 = data_train_gen[data_train_gen.target == 1]\ndf2 = data_train_gen[data_train_gen.target == 0][:NUMBER_OF_1_LABELS]\ndata_train = pd.concat([df1, df2])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f9404c4930b38c69cc30d8161b5b7b18b75cc94"},"cell_type":"code","source":"count_Class=pd.value_counts(data_train.target, sort= True)\ncount_Class.plot(kind = 'pie',  autopct='%1.0f%%')\nplt.title('Pie chart')\nplt.ylabel('')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6e042748985e98e5d612fea7d582082824a00be5"},"cell_type":"markdown","source":"## 1.4 Перемешаем наш сбалансированый data_train_gen"},{"metadata":{"trusted":true,"_uuid":"1b8cc900a89547b8f20ac81517472f56cfa1bc2c"},"cell_type":"code","source":"from sklearn.utils import shuffle\ndata_train = shuffle(data_train)\n# data_train = data_train[:10000]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a27dc7483eb5d5b074ead530ab3955747bf64de6"},"cell_type":"markdown","source":"## 2.1 Рассмотрим самые часто встречающиеся слова в вопросах"},{"metadata":{"trusted":true,"_uuid":"82aa646ad62295b0cf97bbec6ecf2df676f57101"},"cell_type":"code","source":"count1 = Counter(\" \".join(data_train[data_train['target']==0].question_text).split()).most_common(50)\ndf1 = pd.DataFrame.from_dict(count1)\ndf1 = df1.rename(columns={0: \"words_in_excellent \", 1 : \"count_1\"})\ndf1.T","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a3b147ff2a0dd6e8eca0787912d3be070d9868c5"},"cell_type":"markdown","source":"#### Заметим, что самыми популярными топ-50 словами явялются предлоги, местоимения, междометия и прочее, потому уберем такие слова из обучающей выборки , так как они не показывают определенную принадлежность к классу."},{"metadata":{"trusted":true,"_uuid":"28fba0f9a16ae1856676069566010bf35480d62c"},"cell_type":"code","source":"quiestion_words = ['what','when','why','which','who','how', 'whose', 'whome']\nstop_signs = ['.',',',':','...','\\'', '\\\"']\nstop_words = set(stopwords.words('english'))\nstop_words = [w for w in stop_words]\nstop_words = [w for w in stop_words if w not in quiestion_words]   ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6894cfe3ff62921c5fa9a1e0c24e4d57ab453e19"},"cell_type":"markdown","source":"## 2.2 Уберем из предложений самые часто встречающиеся слова "},{"metadata":{"_uuid":"5572d9c3039f7f23079dc0414aba45e1f80e8c56"},"cell_type":"markdown","source":"#### Также приведем все слова к lower case и выкинем некоторые знаки препинания."},{"metadata":{"trusted":true,"_uuid":"ff37a52d361b742da50ed85b4ee6252cd3e0a38c"},"cell_type":"code","source":"cleaned_questions_train = []\n \nfor sentence in data_train['question_text']:\n    new_sentence = [w.lower() for w in word_tokenize(sentence) if not w in stop_words]\n    new_sentence = [w for w in new_sentence if w not in stop_signs]\n         \n    clean = ' '.join(new_sentence)    \n   \n    cleaned_questions_train.append(clean)\n\ncleaned_questions_test = []\nfor sentence in data_test['question_text']:\n    new_sentence = [w.lower() for w in word_tokenize(sentence) if not w in stop_words]\n    new_sentence = [w for w in new_sentence if w not in stop_signs]\n         \n    clean = ' '.join(new_sentence)    \n   \n    cleaned_questions_test.append(clean)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d88e68daa7dea754ed1f7bbd0f3bf7de089067a2"},"cell_type":"code","source":"data_train.insert(loc=0, column=\"debugged_questions\", value=cleaned_questions_train)\ndata_test.insert(loc=0, column=\"debugged_questions\", value=cleaned_questions_test)\ndata_test.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f196339b86c56f382850d9036da20d641be36386"},"cell_type":"markdown","source":"#### Выберем список уникальных слов из train и test, чтобы позже использовать его как словарь для TFidf."},{"metadata":{"trusted":true,"_uuid":"3c8fc81fe0d43849a8e24cffe87317fcc8cabfe4"},"cell_type":"code","source":"words = []\npd.concat([data_train.debugged_questions, data_test.debugged_questions])\nfor sen in [s for s in data_train.question_text]:\n    for word in sen.split():\n        if word not in words:\n            words.append(word)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f6bfed4952ed71112495fac97aefc1cba09f08a"},"cell_type":"code","source":"y = data_train.target","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9956e4e4a99cd1f13c9b52259d9eef2a448c087"},"cell_type":"code","source":"vectorizer = TfidfVectorizer(\"english\", vocabulary = words)\nX = vectorizer.fit_transform(data_train['debugged_questions'])\nX_val = vectorizer.fit_transform(data_test['debugged_questions'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5b81b33cab55a63e4a30508dd38fc4d2fec2b993"},"cell_type":"markdown","source":"## 3.1 Строим решающие деревья по сетке параметров. "},{"metadata":{"trusted":true,"_uuid":"40517386bd879d7d45e3a282686559c260c7d346"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.3, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"adda0b67c273e5d3c3893b36b373792a240c3fb0"},"cell_type":"code","source":"grid = OrderedDict((\n        ('max_depth',np.arange(2,20,4)),\n        ('min_samples_leaf',np.exp(np.linspace(3,8,5)).astype(int))))\n\nresult = {'params':[],'roc_auc_train':[],'roc_auc_valid':[], 'f1_train':[], 'f1_valid':[]}\n\nfor param_values in itertools.product(*grid.values()):\n    param = dict(zip(grid.keys(),param_values))\n    clf = DecisionTreeClassifier(**param)\n\n    clf.fit(X_train,y_train)\n\n    train_pred = clf.predict(X_train)\n    valid_pred = clf.predict(X_test)\n\n    roc_auc_train = roc_auc_score(y_train,train_pred)\n    roc_auc_valid = roc_auc_score(y_test,valid_pred)\n    f1_train = f1_score(y_train,train_pred)\n    f1_test = f1_score(y_test,valid_pred)\n    result['params'].append(param)\n    result['roc_auc_train'].append(roc_auc_train)\n    result['roc_auc_valid'].append(roc_auc_valid)\n    result['f1_train'].append(f1_train)\n    result['f1_valid'].append(f1_test)\n\n# Выводим результаты    \n(pd.DataFrame(result)   \n   .style\n   .background_gradient('Wistia')\n)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"60955b8e497b4c7573d64f2e0c856559a62c64e1"},"cell_type":"markdown","source":"#### видим, что на validation выборке лучший результат по метрике F1 при max_depth = 18 и min_samples_leaf = 20"},{"metadata":{"_uuid":"a58be81f75b07ffd7caff9073291862929cbddde"},"cell_type":"markdown","source":"## 3.2 Возможно найдем лучшие параметры, используя Grid Search"},{"metadata":{"trusted":true,"_uuid":"abcc254fa2f0b167a3f7e47a83c30779bf3a5955"},"cell_type":"code","source":"first_tree = DecisionTreeClassifier()\ntree_params_grid = {'max_depth' : np.arange(2,18,4),'min_samples_leaf': [20,70,244,854]}\ntree_grid=GridSearchCV(first_tree,tree_params_grid,scoring='f1'\n                        ,cv=5,n_jobs = -1)\n\ntree_grid.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bf8dadca1d96477856b101b02faab863ef9a5b45"},"cell_type":"code","source":"tree_grid.best_score_, tree_grid.best_params_","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"21aa781657b6060e7f70408707e651839a866493"},"cell_type":"markdown","source":"#### Параметры, подобранные руками показывают результат(0.76) лучше, чем Grid search(0.75). Тогда окончательную модель строим на параметрах max_depth = 18 и min_samples_leaf = 20"},{"metadata":{"trusted":true,"_uuid":"a4ad707ed253268fcae19443bcde409ec5c17b21"},"cell_type":"code","source":"f_tree = DecisionTreeClassifier(max_depth = 18, min_samples_leaf = 20)\nf_tree.fit(X_train,y_train)\nprediction = f_tree.predict(X_test)\nf1 = f1_score(y_test, prediction)\nf1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a580745c1c20cfc1ae1951cd3bf660e846064466"},"cell_type":"markdown","source":"## Анализ ошибок модели"},{"metadata":{"_uuid":"686e9d172084bef6bc654f2377e75f358fe80032"},"cell_type":"markdown","source":"#### Нарисуем кривую ROC"},{"metadata":{"trusted":true,"_uuid":"e2d31de6fbb5a2b1ae425ca1943a6510f4999586"},"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\n\ntrain_predict = f_tree.predict(X_train)\nvalid_predict = f_tree.predict(X_test)\n\nplt.figure()\nlw = 2\nplt.plot(*roc_curve(y_train,train_predict)[:-1], color='red',\n         lw=lw, label='Train ROC curve (area = %0.3f)' % roc_auc_train)\nplt.plot(*roc_curve(y_test,valid_predict)[:-1], color='darkorange',\n         lw=lw, label='Valid ROC curve (area = %0.3f)' % roc_auc_valid)\nplt.plot([0, 1], [0, 1], color='navy', lw=lw, linestyle='--')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('Receiver operating characteristic')\nplt.legend(loc=\"lower right\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"de47a4bdbee704291c8bd984ee1afea812e0301a"},"cell_type":"markdown","source":"## Делаем submission"},{"metadata":{"trusted":true,"_uuid":"8ffaf162bd8e994ed563b2956283a3a0cc335424"},"cell_type":"code","source":"sub_df = pd.DataFrame({'qid':data_test.qid.values})\nsub_df['prediction'] = f_tree.predict(X_val)\nsub_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"063ff4d7058e2facfcada101f114de1991f42cb5"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cafe60fec134496090e32608ebba6f265d34679f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.5"}},"nbformat":4,"nbformat_minor":1}