{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"quora_train = pd.read_csv(\"../input/train.csv\")\nquora_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"466ed934f194a1913e05665c49647d0433e9bff8"},"cell_type":"code","source":"from nltk.stem import SnowballStemmer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics import f1_score\n\n\ndef stemmatize_sentence(sentence, tokinizer):\n    stemmer = SnowballStemmer('english')\n    return list(map(stemmer.stem, tokinizer(sentence)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0f0415c77e24d026ab001fb30ef288d5e47e80a"},"cell_type":"code","source":"vect = TfidfVectorizer(lowercase=True, max_df=0.8)\ntokenizer = vect.build_tokenizer()\nclf = SGDClassifier(class_weight = \"balanced\", loss='modified_huber', max_iter=2)\nsgd_model = Pipeline([('vect', vect), ('clf', clf)])\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36f786addd57b9942da987636f3d8f97f93d3ae6"},"cell_type":"code","source":"%%time\n\nstemmatized_data = list(map(lambda t: stemmatize_sentence(t, tokenizer), quora_train.question_text.values))\ntrain_X = list(map(lambda t: \" \".join(t), stemmatized_data))\ntrain_y = quora_train.target.values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eda6a1207fd77b98aef4aff07fc829eaca54874d"},"cell_type":"code","source":"np.array(train_X).shape, np.array(train_y).shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec4f2540a163ec6d18aad91cb4de3b109b69cc57"},"cell_type":"code","source":"train_X[4], train_y[4]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"27f981c1b4bc99d7116b4b9dd92c764faf16259e"},"cell_type":"code","source":"%%time\n\nsgd_pred_prob = cross_val_predict(sgd_model, train_X, train_y,\n                          cv=StratifiedKFold(4, random_state=42), n_jobs=1, verbose=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"36d2faaac5eec63640ac4d5e0fba190490efa241"},"cell_type":"code","source":"thresholds = np.arange(0.05, 1, 0.05)\nsgd_pred_arr = []\nfor threshold in thresholds:\n    sgd_pred_arr.append(list(map(lambda x: 1 if x[1]>threshold else 0, sgd_pred_prob)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e5b6a710e7004aabdb20ddb849b1464402b5cfd"},"cell_type":"code","source":"i = 0\nprint(\"SGD with stemming\")\nfor pred in sgd_pred_arr:\n    print(f\"F1-score = {f1_score(train_y, pred)} with threshold = {thresholds[i]}\")\n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2bffed86c58bbaa839b73df053f71afc8b3f5ea6"},"cell_type":"markdown","source":"## Word2Vec"},{"metadata":{"trusted":true,"_uuid":"c7e95e43fff46046a4a11ca2093d8d02917cd9f5"},"cell_type":"code","source":"from gensim.models import KeyedVectors\n\nw2v = KeyedVectors.load_word2vec_format('../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin', binary=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"37c3a784490c85bc75e9fdf57678e67db1b24d60"},"cell_type":"code","source":"w2v[\"Obama\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"53e50c97cc690f8bb2ec3cb09b6e8502b5143772"},"cell_type":"code","source":"w2v.most_similar(positive=['Minsk', 'Russia'], negative=['Belarus'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20f40880f1712cb90dd412a7df6d5512476ddc24"},"cell_type":"code","source":"w2v.most_similar(positive=['Minsk', 'Ukrain'], negative=['Belarus'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d11786bd4b3d9869e30aba8d87a51811963fbc39"},"cell_type":"code","source":"def w2v_sentence(sentence, tokenizer):\n    word_list = tokenizer(sentence)\n    w2v_vector = []\n    for word in word_list:\n        if word in w2v.vocab:\n            w2v_vector.append(w2v[word])\n        else:\n            pass\n    if len(w2v_vector) == 0:\n        return np.zeros(300)\n    else:\n        return np.mean(w2v_vector, axis=0)\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f58fb909aa2b1059aa0f18aff16ae0d397f557ea"},"cell_type":"code","source":"%%time\n\nw2v_data = list(map(lambda t: w2v_sentence(t, tokenizer), quora_train.question_text.values))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"751bdd009db7bfda3161efb145455eaac822f036"},"cell_type":"code","source":"del w2v # чтобы освободить ram, а то кагля умирает :(","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"955b4c165e8358b3f2a670cdd42e5079584a3ac0"},"cell_type":"code","source":"w2v_data[4]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a51a1139b9a79c3a328f37421a2a80400b60db86"},"cell_type":"code","source":"# Воспользуемся той же моделькой\nw2v_pred_prob = cross_val_predict(clf, w2v_data, quora_train.target.values,\n                          cv=StratifiedKFold(4, random_state=42), n_jobs=1, verbose=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"885313d08fab0f3ead91285ccac6a1150b63fe4c"},"cell_type":"code","source":"w2v_pred_arr = []\nfor threshold in thresholds:\n    w2v_pred_arr.append(list(map(lambda x: 1 if x[1]>threshold else 0, w2v_pred_prob)))\n\ni = 0\nprint(\"SGD with w2v\")\nfor pred in w2v_pred_arr:\n    print(f\"F1-score = {f1_score(quora_train.target.values, pred)} with threshold = {thresholds[i]}\")\n    i += 1\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a9ba89d717143338ec7fe72b0fc1e5e4c7034f3"},"cell_type":"markdown","source":"## В целом получилось хуже, но сейчас попробуем делать штуки.\n"},{"metadata":{"trusted":true,"_uuid":"e6bca01ee0ffefac281a93cc550564d1278639b8"},"cell_type":"code","source":"new_X = list(zip(w2v_pred_prob[:,1], sgd_pred_prob[:,1]))\nnew_X[500]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"19552d27c74f15585eaa8144892125eb83f3d020"},"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\n\n\"\"\"log_reg_grid = {\n    'max_iter': [60, 80, 100, 150]\n}\n\nfinal_search = GridSearchCV(LogisticRegression(class_weight='balanced', solver=\"lbfgs\"), log_reg_grid,\n                           cv=StratifiedKFold(4, random_state=42), n_jobs=4, verbose=1,\n                           scoring='f1')\"\"\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b299aa826a1c2493698852d6d88f8e66b85f58eb"},"cell_type":"code","source":"# final_search.fit(new_X, train_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7530fd058d446c8407660da09f171b151612b30"},"cell_type":"code","source":"best_log_reg = LogisticRegression(max_iter=100, class_weight='balanced', solver=\"lbfgs\")\nfinal_pred_prob = cross_val_predict(best_log_reg, new_X, quora_train.target.values,\n                          cv=StratifiedKFold(4, random_state=42), n_jobs=4, verbose=1,\n                          method='predict_proba')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"defdcfd17f94e3b22f9591f06be091dd84e23b8f"},"cell_type":"code","source":"final_pred_arr = []\nfor threshold in thresholds:\n    final_pred_arr.append(list(map(lambda x: 1 if x[1]>threshold else 0, final_pred_prob)))\n\ni = 0\nprint(\"Final model\")\nfor pred in final_pred_arr:\n    print(f\"F1-score = {f1_score(quora_train.target.values, pred)} with threshold = {thresholds[i]}\")\n    i += 1","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"db007306e01dcb028d3cac14c6f021c4ca905551"},"cell_type":"markdown","source":"## Делаем предсказание"},{"metadata":{"trusted":true,"_uuid":"53589b6ed1f9c3244915c58bd67e534fa47c5559"},"cell_type":"code","source":"quora_test = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"264df57a2bf9338a1a0ec57c5816dc5842e43f5e"},"cell_type":"code","source":"sgd_model.fit(train_X, train_y)\nw2v_model = SGDClassifier(class_weight = \"balanced\", loss='modified_huber', max_iter=2)\nw2v_model.fit(w2v_data, train_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5c9465e57c93e7c3652536c432ade6e403d1eafa"},"cell_type":"code","source":"del w2v_data\ndel train_X\ndel train_y\ndel quora_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58fa01dda690632c0741c964c68af15750bc2275"},"cell_type":"code","source":"w2v = KeyedVectors.load_word2vec_format('../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin', binary=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"31e81f7658b99de241d2ebb3cf24fe374cd4a537"},"cell_type":"code","source":"w2v_test = list(map(lambda t: w2v_sentence(t, tokenizer), quora_test.question_text.values))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"067787b081ed862f3da48a5ae0193b7b2a9f07fa"},"cell_type":"code","source":"del w2v","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"70d82ab2625dd5e3272d5eefe470fd317e95fa45"},"cell_type":"code","source":"quora_train = pd.read_csv(\"../input/train.csv\").target.values\nbest_log_reg.fit(new_X, quora_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"038480a23179f60632171d99098825fca9d4e4e9"},"cell_type":"code","source":"del quora_train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7a99847c33460373f1e11f137ad540122bd9317"},"cell_type":"code","source":"test1 = sgd_model.predict_proba(quora_test.question_text.values)[:,1]\ntest2 = w2v_model.predict_proba(w2v_test)[:,1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"333aa657a789c742dc3f42024598888e03589552"},"cell_type":"code","source":"answers_prob = best_log_reg.predict_proba(list(zip(test1, test2)))\nanswers = list(map(lambda x: 1 if x[1]>0.85 else 0, answers_prob))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2cf50d5388753565ecaf8078d6e40a291bd1b8cd"},"cell_type":"code","source":"out_df = pd.DataFrame(np.transpose([quora_test.qid, answers]), columns=[\"qid\", \"prediction\"])\nout_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b80ae807bc5f0247c7b75af4ae0c59531a42f668"},"cell_type":"code","source":"out_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"286b6111dcb5eaf7fcec5f0e12a781e77cbbc71f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}