{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**GIỚI THIỆU VỀ TẬP DỮ LIỆU**\n\nQuora là một nền tảng cho phép mọi người học hỏi lẫn nhau. Trên Quora, mọi người có thể đặt câu hỏi và kết nối với những người khác, những người đóng góp thông tin chi tiết độc đáo và câu trả lời chất lượng. Một thách thức quan trọng là loại bỏ những câu hỏi thiếu chân thành - những câu hỏi được đặt ra dựa trên những tiền đề sai lầm hoặc có ý định đưa ra một tuyên bố hơn là tìm kiếm những câu trả lời hữu ích.\n\nTrong cuộc thi này, Kagglers sẽ phát triển các mô hình xác định và gắn cờ cho các câu hỏi không chân thành.\n\n**Mô tả tệp**\n\n* train.csv - tập huấn luyện\n* test.csv - bộ thử nghiệm\n\n\n**Các trường dữ liệu**\n\n* qid - mã định danh câu hỏi duy nhất\n* question_text - câu hỏi Quora\n* target - câu hỏi có nhãn \"insincere\" có giá trị bằng 1, ngược lại bằng 0","metadata":{}},{"cell_type":"code","source":"import re\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nfrom tqdm import tqdm\ntqdm.pandas()\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom sklearn.metrics import accuracy_score, f1_score\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, GRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, GlobalMaxPooling1D, GlobalAveragePooling1D\nfrom keras.layers import Input, Embedding, Dense, Conv2D, MaxPool2D, concatenate\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D\nfrom keras.optimizers import Adam\nfrom keras.models import Model\nfrom sklearn.model_selection import GridSearchCV, StratifiedKFold\nfrom keras import backend as K\nfrom keras.engine.topology import Layer\nfrom keras import initializers, regularizers, constraints, optimizers, layers\n\n\nfrom keras.layers import *\nfrom keras.models import *\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.initializers import *\nfrom keras.optimizers import *\nimport keras.backend as K\nfrom keras.callbacks import *\nimport tensorflow as tf\nimport os\nimport time\nimport gc\nimport re\nimport glob","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-10T02:35:19.216845Z","iopub.execute_input":"2021-06-10T02:35:19.217163Z","iopub.status.idle":"2021-06-10T02:35:19.235048Z","shell.execute_reply.started":"2021-06-10T02:35:19.217132Z","shell.execute_reply":"2021-06-10T02:35:19.233909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DATA OVERVIEW","metadata":{}},{"cell_type":"markdown","source":"**ĐỌC DỮ LIỆU**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ntest = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:20.526602Z","iopub.execute_input":"2021-06-10T02:35:20.526920Z","iopub.status.idle":"2021-06-10T02:35:24.869600Z","shell.execute_reply.started":"2021-06-10T02:35:20.526891Z","shell.execute_reply":"2021-06-10T02:35:24.868667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:24.871121Z","iopub.execute_input":"2021-06-10T02:35:24.871474Z","iopub.status.idle":"2021-06-10T02:35:25.220924Z","shell.execute_reply.started":"2021-06-10T02:35:24.871437Z","shell.execute_reply":"2021-06-10T02:35:25.218674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Nhận xét:*** Dữ liệu huấn luyện không có giá trị null","metadata":{}},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:25.225767Z","iopub.execute_input":"2021-06-10T02:35:25.226123Z","iopub.status.idle":"2021-06-10T02:35:25.344545Z","shell.execute_reply.started":"2021-06-10T02:35:25.226087Z","shell.execute_reply":"2021-06-10T02:35:25.343611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Nhận xét:*** Dữ liệu kiểm thử không có giá trị null","metadata":{}},{"cell_type":"markdown","source":"**TỈ LỆ PHÂN BỐ NHÃN TRONG TẬP DỮ LIỆU HUẤN LUYỆN**","metadata":{}},{"cell_type":"code","source":"ax, fig = plt.subplots(figsize=(10, 7))\nquestion_class = train[\"target\"].value_counts()\nquestion_class.plot(kind= 'bar', color= [\"blue\", \"orange\"])\nplt.title('Bar chart')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:25.348750Z","iopub.execute_input":"2021-06-10T02:35:25.349094Z","iopub.status.idle":"2021-06-10T02:35:25.701353Z","shell.execute_reply.started":"2021-06-10T02:35:25.349059Z","shell.execute_reply":"2021-06-10T02:35:25.700567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Tỉ lệ phần trăm số câu hỏi Insincere là:\", (len(train.loc[train.target==1])) / (len(train.loc[train.target == 0])) * 100)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:25.702601Z","iopub.execute_input":"2021-06-10T02:35:25.702927Z","iopub.status.idle":"2021-06-10T02:35:25.836468Z","shell.execute_reply.started":"2021-06-10T02:35:25.702890Z","shell.execute_reply":"2021-06-10T02:35:25.834690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Nhận xét***: Số câu hỏi \"insincere\" chỉ chiếm khoảng 6-7% trong tổng số câu hỏi. Dữ liệu bị mất cân bằng khá lớn, do đó độ đo F1 có vẻ thích hợp cho những trường hợp như này","metadata":{}},{"cell_type":"markdown","source":"**PHÂN TÍCH TỪNG CÂU HỎI**","metadata":{}},{"cell_type":"markdown","source":"**Số lượng từ trong câu**","metadata":{}},{"cell_type":"code","source":"words = train['question_text'].apply(lambda x: len(x) - len(''.join(x.split())) + 1)\ntrain['words'] = words\nwords = train.loc[train['words']<200]['words']\nsns.distplot(words, color='g')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:25.837824Z","iopub.execute_input":"2021-06-10T02:35:25.838176Z","iopub.status.idle":"2021-06-10T02:35:33.238614Z","shell.execute_reply.started":"2021-06-10T02:35:25.838135Z","shell.execute_reply":"2021-06-10T02:35:33.237747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Số lượng từ trung bình của các câu hỏi trong dữ liệu huấn luyện là {0:.0f}.'.format(np.mean(train['question_text'].apply(lambda x: len(x.split())))))\nprint('Số lượng từ trung bình của các câu hỏi trong dữ liệu kiểm thử là {0:.0f}.'.format(np.mean(test['question_text'].apply(lambda x: len(x.split())))))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:33.239901Z","iopub.execute_input":"2021-06-10T02:35:33.240400Z","iopub.status.idle":"2021-06-10T02:35:35.360754Z","shell.execute_reply.started":"2021-06-10T02:35:33.240360Z","shell.execute_reply":"2021-06-10T02:35:35.359842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Số lượng từ lớn nhất của các câu hỏi trong dữ liệu huấn luyện là {0:.0f}.'.format(np.max(train['question_text'].apply(lambda x: len(x.split())))))\nprint('Số lượng từ lớn nhất của các câu hỏi trong dữ liệu kiểm thử là {0:.0f}.'.format(np.max(test['question_text'].apply(lambda x: len(x.split())))))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:35.363351Z","iopub.execute_input":"2021-06-10T02:35:35.363713Z","iopub.status.idle":"2021-06-10T02:35:38.235096Z","shell.execute_reply.started":"2021-06-10T02:35:35.363674Z","shell.execute_reply":"2021-06-10T02:35:38.234312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Số lượng ký tự trung bình của các câu hỏi trong dữ liệu huấn luyện là {0:.0f}.'.format(np.mean(train['question_text'].apply(lambda x: len(x)))))\nprint('Số lượng ký tự trung bình của các câu hỏi trong dữ liệu kiểm thử là {0:.0f}.'.format(np.mean(test['question_text'].apply(lambda x: len(x)))))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:38.237494Z","iopub.execute_input":"2021-06-10T02:35:38.238042Z","iopub.status.idle":"2021-06-10T02:35:39.136884Z","shell.execute_reply.started":"2021-06-10T02:35:38.238000Z","shell.execute_reply":"2021-06-10T02:35:39.135998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Nhận xét:*** Có thể thấy độ dài trung bình của các câu hỏi trong tập dữ liệu huấn luyện và kiểm thử tương tự nhau, tuy nhiên có những câu hỏi khá dài trong tập dữ liệu huấn luyện","metadata":{}},{"cell_type":"markdown","source":"# DATA PREPROCESSING","metadata":{}},{"cell_type":"markdown","source":"**Vấn đề:** Như đã phân tích về dữ liệu ở trên, ta sẽ quy chuẩn số lượng từ trong một câu, độ dài của vector sau khi chuẩn hoá câu.","metadata":{}},{"cell_type":"code","source":"embed_size = 300 #độ dài của vector\nmax_features = 100000 #số lượng từ xuất hiện nhiều nhất để huấn luyện\nmaxlen = 80 #số lượng từ trong câu ","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:39.138196Z","iopub.execute_input":"2021-06-10T02:35:39.138706Z","iopub.status.idle":"2021-06-10T02:35:39.143660Z","shell.execute_reply.started":"2021-06-10T02:35:39.138664Z","shell.execute_reply":"2021-06-10T02:35:39.142612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Ở notebook này, ta sẽ không thực hiện tiền xử lý dữ liệu mà chia luôn dữ liệu để huấn luyện**","metadata":{}},{"cell_type":"code","source":"## split to train and val\ntrain, val = train_test_split(train, test_size=0.1, random_state=1023)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:44.967629Z","iopub.execute_input":"2021-06-10T02:35:44.967944Z","iopub.status.idle":"2021-06-10T02:35:45.518120Z","shell.execute_reply.started":"2021-06-10T02:35:44.967911Z","shell.execute_reply":"2021-06-10T02:35:45.517274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Thay các giá trị còn thiếu bằng 'na'**","metadata":{}},{"cell_type":"code","source":"# fill up the missing values\ntrain_X = train[\"question_text\"].fillna(\"_na_\").values\nval_X = val[\"question_text\"].fillna(\"_na_\").values\ntest_X = test[\"question_text\"].fillna(\"_na_\").values","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:47.198928Z","iopub.execute_input":"2021-06-10T02:35:47.199378Z","iopub.status.idle":"2021-06-10T02:35:47.542007Z","shell.execute_reply.started":"2021-06-10T02:35:47.199336Z","shell.execute_reply":"2021-06-10T02:35:47.541180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tokenize các câu hỏi và chuyển thành chuỗi vector**","metadata":{}},{"cell_type":"code","source":"# Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:35:49.056361Z","iopub.execute_input":"2021-06-10T02:35:49.056684Z","iopub.status.idle":"2021-06-10T02:36:37.010825Z","shell.execute_reply.started":"2021-06-10T02:35:49.056656Z","shell.execute_reply":"2021-06-10T02:36:37.009969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Pad chuỗi** - nếu số lượng từ trong câu hỏi lớn hơn 'max_len' thì chuyển thành 'max_len' hoặc nếu số từ trong văn bản ít hơn 'max_len' thì bổ sung thêm số 0 vào các giá trị còn lại","metadata":{}},{"cell_type":"code","source":"# Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:36:37.012246Z","iopub.execute_input":"2021-06-10T02:36:37.012623Z","iopub.status.idle":"2021-06-10T02:36:50.096339Z","shell.execute_reply.started":"2021-06-10T02:36:37.012588Z","shell.execute_reply":"2021-06-10T02:36:50.095473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the target values\ntrain_y = train['target'].values\nval_y = val['target'].values","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:36:50.098082Z","iopub.execute_input":"2021-06-10T02:36:50.098442Z","iopub.status.idle":"2021-06-10T02:36:50.102507Z","shell.execute_reply.started":"2021-06-10T02:36:50.098406Z","shell.execute_reply":"2021-06-10T02:36:50.101672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MODELING","metadata":{}},{"cell_type":"markdown","source":"**Thiết lập model**","metadata":{}},{"cell_type":"code","source":"inp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size)(inp)\nx = Bidirectional(LSTM(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:37:39.350783Z","iopub.execute_input":"2021-06-10T02:37:39.351093Z","iopub.status.idle":"2021-06-10T02:37:40.422008Z","shell.execute_reply.started":"2021-06-10T02:37:39.351063Z","shell.execute_reply":"2021-06-10T02:37:40.421003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Thiết lập model**","metadata":{}},{"cell_type":"code","source":"def train_pred(model, train_X, train_y, val_X, val_y, epochs=2):\n    for e in range(epochs):\n        model.fit(train_X, train_y, batch_size=512, epochs=1, validation_data=(val_X, val_y))\n        pred_val_y = model.predict([val_X], batch_size=1024, verbose=0)\n\n        best_thresh = 0.5\n        best_score = 0.0\n        for thresh in np.arange(0.1, 0.501, 0.01):\n            thresh = np.round(thresh, 2)\n            score = metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n            if score > best_score:\n                best_thresh = thresh\n                best_score = score\n\n        print(\"Val F1 Score: {:.4f}\".format(best_score))\n\n    pred_test_y = model.predict([test_X], batch_size=1024, verbose=0)\n    print('='*100)\n    return pred_val_y, pred_test_y, best_score","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:37:42.890814Z","iopub.execute_input":"2021-06-10T02:37:42.891159Z","iopub.status.idle":"2021-06-10T02:37:42.899323Z","shell.execute_reply.started":"2021-06-10T02:37:42.891130Z","shell.execute_reply":"2021-06-10T02:37:42.898389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train model**","metadata":{}},{"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T02:37:45.158667Z","iopub.execute_input":"2021-06-10T02:37:45.159019Z","iopub.status.idle":"2021-06-10T03:14:11.327010Z","shell.execute_reply.started":"2021-06-10T02:37:45.158987Z","shell.execute_reply":"2021-06-10T03:14:11.326243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_val_y = model.predict([val_X], batch_size=1024, verbose=0)\npred_test_y = model.predict([test_X], batch_size=1024, verbose=0)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T03:14:15.133963Z","iopub.execute_input":"2021-06-10T03:14:15.134327Z","iopub.status.idle":"2021-06-10T03:14:30.632746Z","shell.execute_reply.started":"2021-06-10T03:14:15.134294Z","shell.execute_reply":"2021-06-10T03:14:30.631863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Xác định threshold có F1 cao nhất**","metadata":{}},{"cell_type":"code","source":"def f1_smart(y_true, y_pred):\n    thresholds = []\n    for thresh in np.arange(0.1, 0.501, 0.01):\n        thresh = np.round(thresh, 2)\n        res = f1_score(y_true, (y_pred > thresh).astype(int))\n        thresholds.append([thresh, res])\n        print(\"F1 score at threshold {0} is {1}\".format(thresh, res))\n\n    thresholds.sort(key=lambda x: x[1], reverse=True)\n    best_thresh = thresholds[0][0]\n    best_f1 = thresholds[0][1]\n    print(\"Best threshold: \", best_thresh)\n    return  best_f1, best_thresh","metadata":{"execution":{"iopub.status.busy":"2021-06-10T03:14:30.636274Z","iopub.execute_input":"2021-06-10T03:14:30.636537Z","iopub.status.idle":"2021-06-10T03:14:30.642078Z","shell.execute_reply.started":"2021-06-10T03:14:30.636512Z","shell.execute_reply":"2021-06-10T03:14:30.641302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1, threshold = f1_smart(val_y, pred_val_y)\nprint('Optimal F1: {} at threshold: {}'.format(f1, threshold))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T03:14:30.643472Z","iopub.execute_input":"2021-06-10T03:14:30.644027Z","iopub.status.idle":"2021-06-10T03:14:32.342679Z","shell.execute_reply.started":"2021-06-10T03:14:30.643995Z","shell.execute_reply":"2021-06-10T03:14:32.341850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Nhận xét:** F1 score cao nhất = 0.64 với threshold = 0.37","metadata":{}},{"cell_type":"markdown","source":"**Tạo submission**","metadata":{}},{"cell_type":"code","source":"pred_test_y = (pred_test_y >threshold).astype(int)\nout_df = pd.DataFrame({\"qid\":test[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T03:15:18.670144Z","iopub.execute_input":"2021-06-10T03:15:18.670487Z","iopub.status.idle":"2021-06-10T03:15:19.819426Z","shell.execute_reply.started":"2021-06-10T03:15:18.670456Z","shell.execute_reply":"2021-06-10T03:15:19.818560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}