{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom gensim.models import KeyedVectors\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nvect = TfidfVectorizer()\nsklearn_tokenizer = vect.build_tokenizer()\n\ndf = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntokenized = [sklearn_tokenizer(sent) for sent in df.question_text]\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-19T02:22:47.355751Z","iopub.execute_input":"2021-12-19T02:22:47.356616Z","iopub.status.idle":"2021-12-19T02:23:06.848248Z","shell.execute_reply.started":"2021-12-19T02:22:47.356498Z","shell.execute_reply":"2021-12-19T02:23:06.847180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2021-12-19T01:40:05.138515Z","iopub.execute_input":"2021-12-19T01:40:05.138865Z","iopub.status.idle":"2021-12-19T01:40:05.149763Z","shell.execute_reply.started":"2021-12-19T01:40:05.138821Z","shell.execute_reply":"2021-12-19T01:40:05.148797Z"}}},{"cell_type":"code","source":"w2v = KeyedVectors.load_word2vec_format('/kaggle/input/googlenewsvectors/GoogleNews-vectors-negative300-SLIM.bin', binary=True)","metadata":{"execution":{"iopub.status.busy":"2021-12-19T01:40:25.81968Z","iopub.execute_input":"2021-12-19T01:40:25.819957Z","iopub.status.idle":"2021-12-19T01:40:35.234864Z","shell.execute_reply.started":"2021-12-19T01:40:25.819928Z","shell.execute_reply":"2021-12-19T01:40:35.233961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_w2v_vect(token, w2v):\n    token = token.lower()\n\n    if token in w2v.index_to_key:\n        return w2v[token]\n    else:\n        return np.zeros(w2v.vector_size)\n    \ndef get_features_from_text(tokenized_sent, w2v):\n    vect = np.array([get_w2v_vect(token, w2v) \n                    for token in tokenized_sent]).mean(axis=0)\n  \n    vect_norm = np.linalg.norm(vect)\n    if vect_norm != 0:\n        return vect/vect_norm\n    else:\n        return vect\n    \nw2v_data = np.array([get_features_from_text(tokenized_sent, w2v) \n                    for tokenized_sent in tokenized[::5]])","metadata":{"execution":{"iopub.status.busy":"2021-12-18T21:12:59.956001Z","iopub.execute_input":"2021-12-18T21:12:59.95667Z","iopub.status.idle":"2021-12-18T21:47:32.982838Z","shell.execute_reply.started":"2021-12-18T21:12:59.956633Z","shell.execute_reply":"2021-12-18T21:47:32.98163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df.target[::5]","metadata":{"execution":{"iopub.status.busy":"2021-12-18T21:47:32.984306Z","iopub.execute_input":"2021-12-18T21:47:32.98458Z","iopub.status.idle":"2021-12-18T21:47:32.989988Z","shell.execute_reply.started":"2021-12-18T21:47:32.984552Z","shell.execute_reply":"2021-12-18T21:47:32.989288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import SGDClassifier\nclf = SGDClassifier(loss='modified_huber', penalty='elasticnet', l1_ratio=0.1, alpha=1e-6, \n                    shuffle=True, class_weight={0: y.mean(), 1: 1-y.mean()}, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2021-12-18T21:47:32.991619Z","iopub.execute_input":"2021-12-18T21:47:32.992188Z","iopub.status.idle":"2021-12-18T21:47:33.198907Z","shell.execute_reply.started":"2021-12-18T21:47:32.992148Z","shell.execute_reply":"2021-12-18T21:47:33.197708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_predict, StratifiedKFold, RandomizedSearchCV\npreds = cross_val_predict(clf, w2v_data, y, cv=StratifiedKFold(5), \n                          n_jobs=1, method='predict_proba')","metadata":{"execution":{"iopub.status.busy":"2021-12-18T21:49:19.796123Z","iopub.execute_input":"2021-12-18T21:49:19.796743Z","iopub.status.idle":"2021-12-18T21:51:56.191172Z","shell.execute_reply.started":"2021-12-18T21:49:19.796673Z","shell.execute_reply":"2021-12-18T21:51:56.189998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score, classification_report, f1_score","metadata":{"execution":{"iopub.status.busy":"2021-12-18T22:26:05.248531Z","iopub.execute_input":"2021-12-18T22:26:05.249573Z","iopub.status.idle":"2021-12-18T22:26:05.253616Z","shell.execute_reply.started":"2021-12-18T22:26:05.249512Z","shell.execute_reply":"2021-12-18T22:26:05.252789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_target = f1_score(y,preds[:,1]>=0.8)","metadata":{"execution":{"iopub.status.busy":"2021-12-18T22:33:38.573275Z","iopub.execute_input":"2021-12-18T22:33:38.573611Z","iopub.status.idle":"2021-12-18T22:33:38.673219Z","shell.execute_reply.started":"2021-12-18T22:33:38.57358Z","shell.execute_reply":"2021-12-18T22:33:38.672112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame({\"qid\":df[\"qid\"], \"prediction\":pred_target}).to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-12-19T00:31:15.72968Z","iopub.execute_input":"2021-12-19T00:31:15.730279Z","iopub.status.idle":"2021-12-19T00:31:15.839149Z","shell.execute_reply.started":"2021-12-19T00:31:15.730129Z","shell.execute_reply":"2021-12-19T00:31:15.838171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}