{"cells":[{"metadata":{"trusted":true,"_uuid":"416a52ea4f0a739c0bc4bbe8b8f85ec8b218bd65"},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport nltk\nfrom nltk.corpus import stopwords \n\nfrom nltk.corpus import stopwords \nfrom nltk.tokenize import word_tokenize \n\n\nfrom nltk.stem import WordNetLemmatizer\nwordnet_lemmatizer = WordNetLemmatizer()\n\nimport gensim\n\nfrom gensim import models, corpora\n\nfrom sklearn.decomposition import TruncatedSVD\n\nfrom sklearn.tree import DecisionTreeClassifier","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d17ea5d9d2943f5e31d8c7fb03a431e22c57a06"},"cell_type":"code","source":"data_train_gen = pd.read_csv(\"../input/train.csv\")\ndata_test = pd.read_csv(\"../input/test.csv\")\nprint(\"Train datasets shape:\", data_train_gen.shape)\nprint(\"Test datasets shape:\", data_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"306f518c2c3eacb6a053111ff47a67e92d91df09"},"cell_type":"markdown","source":"# 1.1 Сделаю нужное соотношение классов, выкину stop_words, сделаю леммтизацию."},{"metadata":{"trusted":true,"_uuid":"117615f69ed958032d5076b7bf958c8527b461e9"},"cell_type":"code","source":"num = 653826\nbad_df = data_train_gen[data_train_gen.target == 1]\ngood_df = data_train_gen[data_train_gen.target == 0][:num]\ndata_train = pd.concat([bad_df, good_df])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a8a193b2cdf5b066b9a9c232a141812c37dd776"},"cell_type":"code","source":"from sklearn.utils import shuffle\ndata_train = shuffle(data_train)\ndata_train = data_train[:136520]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e214657d7433e9bc2e170ef66691b0f7f301486"},"cell_type":"code","source":"quiestion_words = ['what','when','why','which','who','how', 'whose', 'whome', 'people', 'i', \n                  'n\\'t','\\'s','like','get','would','would']\nstop_signs = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√']\nstop_words = stopwords.words('english')\n# stop_words = stop_words.extend(quiestion_words)\nfor w in quiestion_words:\n    stop_words.append(w)\nfor w in stop_signs:\n    stop_words.append(w)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aecc13520352b409ea2fc21600add2d8486a5311"},"cell_type":"code","source":"cleaned_questions_train = []\n \nfor sentence in data_train['question_text']:\n    new_sentence = [wordnet_lemmatizer.lemmatize(w).lower() for w in word_tokenize(sentence)]\n    new_sentence = [w for w in new_sentence if w not in stop_words]\n    new_sentence = [w for w in new_sentence if len(w)>3]\n         \n    clean = ' '.join(new_sentence)    \n   \n    cleaned_questions_train.append(clean)\n\ncleaned_questions_test = []\nfor sentence in data_test['question_text']:\n    new_sentence = [wordnet_lemmatizer.lemmatize(w).lower() for w in word_tokenize(sentence)]\n    new_sentence = [w for w in new_sentence if w not in stop_words]\n    new_sentence = [w for w in new_sentence if len(w)>3]\n         \n    clean = ' '.join(new_sentence)    \n   \n    cleaned_questions_test.append(clean)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d18a5b312ebe8119544dcdcd46416baf2049c2ff"},"cell_type":"code","source":"data_train.insert(loc=0, column=\"debugged_questions\", value=cleaned_questions_train)\ndata_test.insert(loc=0, column=\"debugged_questions\", value=cleaned_questions_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1fa0b605d593a4e413c9b07e56d4f44216ce6217"},"cell_type":"code","source":"data_train.fillna('', inplace = True)\ndata_test.fillna('', inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0a8f324ef8d521f48ed255b8dc7cffe36900453b"},"cell_type":"code","source":"train = data_train.ix[:, (1,0, 3)]\ntest = data_test.ix[:, (1,0)]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9f972d6bd62114c365dc97a1381252b99b8763db"},"cell_type":"markdown","source":"# 1.2 Получу word2vec представление"},{"metadata":{"trusted":true,"_uuid":"d8c1148c47a5aa54cf5390460deff4f7986c62f0"},"cell_type":"code","source":"train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7662a490b68d5fb80917e27726e6d66e0ab55e0a"},"cell_type":"code","source":"train['tokens_list'] = train.debugged_questions.apply(lambda x: x.split())\ntest['tokens_list'] = test.debugged_questions.apply(lambda x: x.split())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"968557e39ce9a6b97fe21ceea7837d1397ce5e85"},"cell_type":"code","source":"word2vec_path = \"../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin\"\nword2vec = gensim.models.KeyedVectors.load_word2vec_format(word2vec_path, binary=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0fd05fadbd41b9f178f776e8bbb99f6b774fbc0c"},"cell_type":"code","source":"len(test.tokens_list)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"96d4b182527899bef63dd2e76d680859e81ab404"},"cell_type":"code","source":"def get_average_word2vec(tokens_list, vector, generate_missing=False, k=300):\n    if len(tokens_list)<1:\n        return np.zeros(k)\n    if generate_missing:\n        vectorized = [vector[word] if word in vector else np.random.rand(k) for word in tokens_list]\n    else:\n        vectorized = [vector[word] if word in vector else np.zeros(k) for word in tokens_list]\n    length = len(vectorized)\n    summed = np.sum(vectorized, axis=0)\n    averaged = np.divide(summed, length)\n    return averaged\n\ndef get_word2vec_embeddings(vectors, clean_questions, generate_missing=False):\n    embeddings = clean_questions['tokens_list'].apply(lambda x: get_average_word2vec(x, vectors, \n                                                                                generate_missing=generate_missing))\n    return embeddings.tolist()\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5428e7796ad28ec4c2cde6e6279b2a264bc53b84"},"cell_type":"code","source":"test_embeddings = get_word2vec_embeddings(word2vec, test)\ntrain_embeddings = get_word2vec_embeddings(word2vec, train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"107ee4f7c3a89fc96ce3166db571de0963b3726d"},"cell_type":"code","source":"test_embeddings = pd.DataFrame(test_embeddings)\ntrain_embeddings = pd.DataFrame(train_embeddings)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"54c30b12f509d619f9aac8408e21ccb791371a07"},"cell_type":"markdown","source":"# 2.1 Делаю модель LDA"},{"metadata":{"trusted":true,"_uuid":"4cb81976f6e5577acf8d7e4fd811452953c935ed"},"cell_type":"code","source":"# ну или импортирую заранее обученную\n# ldamodel = models.ldamodel.LdaModel.load(\"ldamodel3_lkcd\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"92216b7350c38cd1b905e9cd908e1a7f2f28d5a8"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6023e2201eaa7352890be9af332b4119578fe9e3"},"cell_type":"code","source":"gen = pd.concat([train[train.columns[[0,1,3]]], test], axis = 0)\ntokens = gen.tokens_list.tolist()\ndictionary = corpora.Dictionary(tokens)\ncorpus = []\nfor token in tokens: corpus.append(dictionary.doc2bow(token))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c56e16f4aa78f2d9cb29cc05bd89e4a1a5606c3a"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4732c5cb1bd5544e955af7969835ea008be7eec0"},"cell_type":"code","source":"tetok = test.tokens_list.tolist()\ntrtok = train.tokens_list.tolist()\n# wtok = weird_tok.tolist()\ntecor = []\ntrcor = []\n# wecor = []\nfor token in tetok: tecor.append(dictionary.doc2bow(token))\nfor token in trtok: trcor.append(dictionary.doc2bow(token))\n# for token in wtok: wecor.append(dictionary2.doc2bow(token))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"81a867533673b499c0598851271a50e6c27edd52"},"cell_type":"code","source":"np.random.seed(76543)\n%time ldamodel = models.ldamodel.LdaModel(corpus, id2word=dictionary, num_topics=200, passes=3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a2e3890a86a43ace30559d6d9c9423d8548c0b74"},"cell_type":"markdown","source":"# 2.2 Сделаю матрицу распределения тем по документам, чтобы потом ее добавить к изходным выборкам."},{"metadata":{"trusted":true,"_uuid":"354f7decd2e9ca50d4b7c7a08201138918856aaf"},"cell_type":"code","source":"test_topics = ldamodel.get_document_topics(tecor)\ntrain_topics = ldamodel.get_document_topics(trcor)\n# weird_topics = ldamodel.get_document_topics(wecor)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d9e5384246137edc0a8766fd3cd083dbeaee8da1"},"cell_type":"code","source":"nums1 = []\nfor list in [ldamodel.get_document_topics(corp) for corp in trcor]:\n    num1 = []\n    for tup in list:\n        num1.append(tup)\n    nums1.append(num1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ccf2849aa9daebc68466792b5726b58a46e70cc6"},"cell_type":"code","source":"nums2 = []\nnn = []\nfor list in [ldamodel.get_document_topics(corp) for corp in tecor]:\n    num2 = []\n    for tup in list:\n        num2.append(tup)\n    nums2.append(num2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"48a1ffe33cca6b81ab649467e5046e919f2f141c"},"cell_type":"code","source":"X_ps_tr = np.ndarray([len(trcor), 200])\nX_ps_te = np.ndarray([len(tecor), 200])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"199dc45df4d81d3860d315e8cbf95740124c753d"},"cell_type":"code","source":"for i in range(len(nums1)):\n    if i == 136523: pass\n    for j in range(len(nums1[i])):\n        X_ps_tr[i, nums1[i][j][0]] = nums1[i][j][1]\n        \n        \nfor i in range(len(nums2)):\n    for j in range(len(nums2[i])):\n        X_ps_te[i, nums2[i][j][0]] = nums2[i][j][1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"75db05d33cd6add3a662523179cb4a6e54907737"},"cell_type":"code","source":"X_ps_te = pd.DataFrame(X_ps_te)\nX_ps_tr = pd.DataFrame(X_ps_tr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b8a4e5ec7ab1466adeba065f88f7e6d92e613e61"},"cell_type":"code","source":"train.shape, train_embeddings.shape, X_ps_tr.shape, test.shape, test_embeddings.shape, X_ps_te.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db25947ca777993fe758660bc1aec6d294386fa1"},"cell_type":"code","source":"train.index = pd.RangeIndex(0, train.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b28b2f38105f342dbc2bda0c4c0160fccf4b4b3"},"cell_type":"code","source":"trainig = pd.concat([train.qid, train.target, train_embeddings, X_ps_tr], axis = 1)\ntesting = pd.concat([test.qid, test_embeddings, X_ps_te], axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"11b8eab615203a70dd1c731a610f8ea6cf413766"},"cell_type":"code","source":"trainig.shape, testing.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d7a0f189ec8945f61f6054971632ae050699d6b"},"cell_type":"code","source":"testing.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"00951e173fe7a686d1ebdafdbb788ed203fc1df2"},"cell_type":"code","source":"lsa = TruncatedSVD(n_components=100)\nlsa.fit(trainig[trainig.columns[2:]])\nlsa_scores_train = lsa.transform(trainig[trainig.columns[2:]])\nlsa_scores_test = lsa.transform(testing[testing.columns[1:]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"80320ec83d689eaae361726f84f0975e3e2dec2d"},"cell_type":"code","source":"lsa_scores_train = pd.DataFrame(lsa_scores_train)\nlsa_scores_test = pd.DataFrame(lsa_scores_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"585574dc3545ab57a4baa83b5e8c0fe38ad5ff57"},"cell_type":"code","source":"y = trainig.target","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e194d05ea6796e44ee2a5d44a4c75731f80a267b"},"cell_type":"code","source":"clf = DecisionTreeClassifier(max_depth = 12, min_samples_leaf = 20)\nclf.fit(lsa_scores_train,y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6de1383809523a6100dcfa03bcb3347e417253a4"},"cell_type":"code","source":"sub_df = pd.DataFrame({'qid':testing.qid.values})\nsub_df['prediction'] = clf.predict(lsa_scores_test)\nsub_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3265bd9d3f3a71567586ed982a492744e1d0e50a"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a740080bf9789a49da98ce9eb621044e996d542d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b4d99083c0765dadf73bd4937b994b946b0c4be"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.5"}},"nbformat":4,"nbformat_minor":1}