{"cells":[{"metadata":{"_uuid":"7aeae52bc2d3220ca2ed1b645d6655b397524b01"},"cell_type":"markdown","source":"## 1.1 Загрузка необходимых библиотек"},{"metadata":{"trusted":true,"_uuid":"1dd96912b7ac935b29a01a2a1e62dad6b8bff70a"},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\n\nfrom nltk.corpus import stopwords \nfrom nltk.tokenize import word_tokenize \n\nimport nltk\nnltk.download('stopwords')\nnltk.download('punkt')\n\nfrom collections import Counter\nimport xgboost as xgb\nfrom xgboost import XGBClassifier\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.metrics import accuracy_score\nfrom sklearn import feature_extraction, model_selection, naive_bayes, metrics, svm\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.metrics import f1_score\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import make_scorer\nfrom collections import OrderedDict\nfrom sklearn.metrics import roc_auc_score, log_loss\nimport itertools","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"915ffdf9eecaf01aac74849cf11b5661489a4b7f"},"cell_type":"code","source":"data_train_gen = pd.read_csv(\"../input/train.csv\")\ndata_test = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e8a5da79ab8b62cff792e275d402af759823ec6"},"cell_type":"code","source":"data_train_gen.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe8a766034373e4c7de4713f786750d80e416676"},"cell_type":"code","source":"data_train_gen.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f769e872a8a2536280179636e566086901485eac"},"cell_type":"markdown","source":"##  1.2 Посмотрим на число вхождений"},{"metadata":{"trusted":true,"_uuid":"a8ede9d284be5ea1741d3a269630799a9f926a59"},"cell_type":"code","source":"NUMBER_OF_1_LABELS = data_train_gen.target.value_counts()[1]\nNUMBER_OF_1_LABELS","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"451272b2535c04999cc9e45152f7003d7bf6b4b5"},"cell_type":"code","source":"count_Class=pd.value_counts(data_train_gen.target, sort= True)\ncount_Class.plot(kind= 'bar', color= [\"blue\"])\nplt.title('Bar chart')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"232087f6e194b513fd8335e19c8be49077a40e93"},"cell_type":"code","source":"count_Class.plot(kind = 'pie',  autopct='%1.0f%%')\nplt.title('Pie chart')\nplt.ylabel('')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c1bc1d1840a8969d38c686720edc867a004afaf8"},"cell_type":"markdown","source":"## 1.3 Сбалансируем выборку"},{"metadata":{"trusted":true,"_uuid":"272eec7eaca57be46403a2741575e39239bb8c78"},"cell_type":"code","source":"df1 = data_train_gen[data_train_gen.target == 1]\ndf2 = data_train_gen[data_train_gen.target == 0][:NUMBER_OF_1_LABELS]\ndata_train = pd.concat([df1, df2])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"94ea958160776dd63164c212eead8d032fd8d9b5"},"cell_type":"code","source":"count_Class=pd.value_counts(data_train.target, sort= True)\ncount_Class.plot(kind = 'pie',  autopct='%1.0f%%')\nplt.title('Pie chart')\nplt.ylabel('')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7d4d4280f967a64d61597652673a33680c7cd11b"},"cell_type":"markdown","source":"## 1.4 Перемешаем наш сбалансированый data_train_gen"},{"metadata":{"trusted":true,"_uuid":"3549f86088cb9c8ba0f2b3c5f2dc619d09604ed0"},"cell_type":"code","source":"from sklearn.utils import shuffle\ndata_train = shuffle(data_train)\ndata_train = data_train[:10000]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bfbd40b518873be305625af739f66748a0544c02"},"cell_type":"code","source":"words = []\nfor sen in [s for s in data_train.question_text]:\n    for word in sen.split():\n        if word not in words:\n            words.append(word)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7385c40490abb551130a286635d3ee718bf60cde"},"cell_type":"code","source":"len(words), len(pd.Series(words).unique())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2fb18790326801cc660207a36fafb3888c07bbd9"},"cell_type":"markdown","source":"## 2.1 Рассмотрим самые часто встречающиеся слова в вопросах"},{"metadata":{"trusted":true,"_uuid":"ae28dc4e1516bdfa6b92023129eba3429fe01c14"},"cell_type":"code","source":"count1 = Counter(\" \".join(data_train[data_train['target']==0].question_text).split()).most_common(50)\ndf1 = pd.DataFrame.from_dict(count1)\ndf1 = df1.rename(columns={0: \"words_in_excellent \", 1 : \"count_1\"})\ndf1.T","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"84ca282acef040d86cc6eb4eabc3394bbf75a78e"},"cell_type":"markdown","source":"#### Заметим, что самыми популярными топ-50 словами явялются предлоги, местоимения, междометия и прочее, потому уберем такие слова из обучающей выборки , так как они не показывают определенную принадлежность к классу."},{"metadata":{"trusted":true,"_uuid":"19f13e21913f8bc237a3d54343d2d85fe9165fc4"},"cell_type":"code","source":"quiestion_words = ['what','when','why','which','who','how', 'whose', 'whome']\nstop_signs = ['.',',',':','...','\\'', '\\\"']\nstop_words = set(stopwords.words('english'))\nstop_words = [w for w in stop_words]\nstop_words = [w for w in stop_words if w not in quiestion_words]   ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"71953135175a184be4937dd4d00e6419bae058e2"},"cell_type":"markdown","source":"## 2.2 Уберем из предложений самые часто встречающиеся слова "},{"metadata":{"_uuid":"8e1ff10ab546cadeb705e3049868c2bd4891c159"},"cell_type":"markdown","source":"#### Также приведем все слова к lower case и выкинем некоторые знаки препинания."},{"metadata":{"trusted":true,"_uuid":"26741aee89e9c14d0b6ca4ef5b35b1c60f04fc5c"},"cell_type":"code","source":"cleaned_questions_train = []\n \nfor sentence in data_train['question_text']:\n    new_sentence = [w.lower() for w in word_tokenize(sentence) if not w in stop_words]\n    new_sentence = [w for w in new_sentence if w not in stop_signs]\n         \n    clean = ' '.join(new_sentence)    \n   \n    cleaned_questions_train.append(clean)\n\ncleaned_questions_test = []\nfor sentence in data_test['question_text']:\n    new_sentence = [w.lower() for w in word_tokenize(sentence) if not w in stop_words]\n    new_sentence = [w for w in new_sentence if w not in stop_signs]\n         \n    clean = ' '.join(new_sentence)    \n   \n    cleaned_questions_test.append(clean)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9aa787db0cbdfaa71ee581e3d4fa9f3e26ac7da7"},"cell_type":"code","source":"data_train.insert(loc=0, column=\"debugged_questions\", value=cleaned_questions_train)\ndata_test.insert(loc=0, column=\"debugged_questions\", value=cleaned_questions_test)\ndata_test.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d94fad66eb54b23091b52417593cbfd29355ab07"},"cell_type":"markdown","source":"#### Выберем список уникальных слов из train и test, чтобы позже использовать его как словарь для TFidf."},{"metadata":{"trusted":true,"_uuid":"e9c6171c5a14b0e57493e8671b970cc30511630f"},"cell_type":"code","source":"words = []\npd.concat([data_train.debugged_questions, data_test.debugged_questions])\nfor sen in [s for s in data_train.question_text]:\n    for word in sen.split():\n        if word not in words:\n            words.append(word)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b1d57dfdd26ffd82aa09dbc1a1aef60cd3cddcc"},"cell_type":"code","source":"y = data_train.target","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"270b82e40771fb70fbd61b51b52b90eeeb715ab4"},"cell_type":"code","source":"vectorizer = TfidfVectorizer(\"english\", vocabulary = words)\nX = vectorizer.fit_transform(data_train['debugged_questions'])\nX_val = vectorizer.fit_transform(data_test['debugged_questions'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"36f3cf4407371d0b88ae40653f9a6d0f18c27c42"},"cell_type":"markdown","source":"## 2.3 Train test split"},{"metadata":{"trusted":true,"_uuid":"5c824b7278cb8dad4962cebfd6bfb7d3dbf971a4"},"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.3, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6332d59fa1abbcd47b81be286a341dc02223d211"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a5a4bcfad1ba51d2273be039d1a2bb6d0f6f047f"},"cell_type":"code","source":"clf = GradientBoostingClassifier(learning_rate=0.1, min_samples_split=500, max_depth=8, max_features='sqrt', subsample=0.8)\nclf.fit(X_train, y_train)\npredictions = clf.predict(X_test)\nprint('F-Score: {}'.format(f1_score(y_test, predictions, average='macro')))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c30001507b7bad0ddee5fdaa34ac792906638fcd"},"cell_type":"code","source":"\n\ngrid = OrderedDict((\n        ('max_depth',np.arange(2,20,4)),\n        ('min_samples_leaf',np.exp(np.linspace(3,8,5)).astype(int))\n        \n\n       ))\n\nresult = {'params':[],'roc_auc_train':[],'roc_auc_valid':[], 'f1_train':[], 'f1_valid':[]}\n\nfor param_values in itertools.product(*grid.values()):\n    param = dict(zip(grid.keys(),param_values))\n    clf = DecisionTreeClassifier(**param)\n\n    clf.fit(X_train,y_train)\n\n    train_pred = clf.predict(X_train)\n    valid_pred = clf.predict(X_test)\n\n    roc_auc_train = roc_auc_score(y_train,train_pred)\n    roc_auc_valid = roc_auc_score(y_test,valid_pred)\n    f1_train = f1_score(y_train,train_pred)\n    f1_test = f1_score(y_test,valid_pred)\n    result['params'].append(param)\n    result['roc_auc_train'].append(roc_auc_train)\n    result['roc_auc_valid'].append(roc_auc_valid)\n    result['f1_train'].append(f1_train)\n    result['f1_valid'].append(f1_test)\n\n# Выводим результаты    \n(pd.DataFrame(result)   \n   .style\n   .background_gradient('Wistia')\n)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"35c32d2a68f49e8243951acf3cf4a8f96b894b22"},"cell_type":"markdown","source":"#### видим, что на validation выборке лучший результат по метрике F1 при max_depth = 10 и min_samples_leaf = 20"},{"metadata":{"trusted":true,"_uuid":"97e6212dfef6fa370b1489b3efa77ae1dd6aceca"},"cell_type":"code","source":"{'max_depth': 10, 'max_features': 28, 'min_samples_leaf': 20}\nf_tree = DecisionTreeClassifier(max_depth = 6, min_samples_leaf = 20)\nf_tree.fit(X_train,y_train)\nprediction = f_tree.predict(X_test)\nf1 = f1_score(y_test, prediction)\nf1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fd110e83e4f4bb0df01638fcc75c19652ce5b8ca"},"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\nfirst_tree = DecisionTreeClassifier()\ntree_params_grid = {'max_depth' : np.arange(2,18,4),'min_samples_leaf': [20,70,244,854]}\ntree_grid=GridSearchCV(first_tree,tree_params_grid,scoring='f1'\n                        ,cv=5,n_jobs = -1)\n\ntree_grid.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b73daf85f9301f16791b6ba2079d8467901b000"},"cell_type":"code","source":"tree_grid.best_score_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3eefd8276e2c87c46abba37d94940423ee470130"},"cell_type":"code","source":"tree_grid.best_params_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"696e9dd13757eaa8f1fe0f0794f9d6e5da4f9113"},"cell_type":"code","source":"{'max_depth': 10, 'max_features': 28, 'min_samples_leaf': 20}\nf_tree = DecisionTreeClassifier(max_depth = 6, min_samples_leaf = 20)\nf_tree.fit(X_train,y_train)\nprediction = f_tree.predict(X_test)\nf1 = f1_score(y_test, prediction)\nf1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36cb1e6fcb32019f39563aa381c319901fa7052f"},"cell_type":"code","source":"sub_df = pd.DataFrame({'qid':data_test.qid.values})\nsub_df['prediction'] = f_tree.predict(X_val)\nsub_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c07ad6f5f7ddfc8f86726d224ddc44a13ccac65c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}