{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Quora Insincere Questions**","metadata":{}},{"cell_type":"markdown","source":"### **Giới thiệu:**\n​\nQuora là một nền tảng cho phép mọi người học hỏi lẫn nhau. Trên Quora, mọi người có thể đặt câu hỏi và kết nối với những người khác, những người đưa ra những thông tin bổ ích và câu trả lời chất lượng. Có một thách thức là loại bỏ những câu hỏi thiếu chân thành - những câu hỏi được đặt ra dựa trên ý đồ sai hoặc có ý định đưa ra một tuyên bố hơn là tìm những câu trả lời hữu ích.\n​\n\nBài toán: Cho một bộ dữ liệu gồm những câu hỏi chân thành (sincere questions) và những câu hỏi không chân thành (insincere questions), cần xác định và đánh dấu những câu hỏi không chân thành.\n​","metadata":{}},{"cell_type":"code","source":"import os\nimport json\nimport string\nimport numpy as np\nimport pandas as pd\nfrom pandas.io.json import json_normalize\nimport matplotlib.pyplot as plt\nimport seaborn as sns\ncolor = sns.color_palette()\n\n%matplotlib inline\n\nfrom wordcloud import WordCloud, STOPWORDS\nfrom collections import defaultdict #Tương tự như python dictionary nhưng sẽ tự động sinh ra một default key khi truy cập key không tồn tại \n\nfrom plotly import tools\nimport plotly.offline as py\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, f1_score, confusion_matrix\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import LinearSVC\nfrom sklearn.naive_bayes import MultinomialNB\n\nfrom imblearn.over_sampling import RandomOverSampler\nfrom imblearn.under_sampling import RandomUnderSampler\n\npd.options.mode.chained_assignment = None\npd.options.display.max_columns = 999","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-11T02:44:12.170935Z","iopub.execute_input":"2021-06-11T02:44:12.171604Z","iopub.status.idle":"2021-06-11T02:44:13.772931Z","shell.execute_reply.started":"2021-06-11T02:44:12.171462Z","shell.execute_reply":"2021-06-11T02:44:13.771730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **1. Phân tích dữ liệu** ","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:44:13.775580Z","iopub.execute_input":"2021-06-11T02:44:13.776174Z","iopub.status.idle":"2021-06-11T02:44:19.860380Z","shell.execute_reply.started":"2021-06-11T02:44:13.776077Z","shell.execute_reply":"2021-06-11T02:44:19.859274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:44:19.862445Z","iopub.execute_input":"2021-06-11T02:44:19.862872Z","iopub.status.idle":"2021-06-11T02:44:19.891515Z","shell.execute_reply.started":"2021-06-11T02:44:19.862829Z","shell.execute_reply":"2021-06-11T02:44:19.890674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:44:19.893015Z","iopub.execute_input":"2021-06-11T02:44:19.893303Z","iopub.status.idle":"2021-06-11T02:44:19.899130Z","shell.execute_reply.started":"2021-06-11T02:44:19.893275Z","shell.execute_reply":"2021-06-11T02:44:19.898229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Bộ dữ liệu train có tổng cộng 1306122 hàng và 3 cột\n","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:44:19.900505Z","iopub.execute_input":"2021-06-11T02:44:19.900806Z","iopub.status.idle":"2021-06-11T02:44:20.168948Z","shell.execute_reply.started":"2021-06-11T02:44:19.900776Z","shell.execute_reply":"2021-06-11T02:44:20.167938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dữ liệu không có giá trị null.\n\nTập dữ liệu train gồm 2 cột kiểu text (qid và question_text) và 1 cột kiểu số (target):\n\n* qid: ID của câu hỏi.\n* question_text: Nội dung của câu hỏi.\n* target: insincere = 1, sincere = 0","metadata":{}},{"cell_type":"code","source":"val_counts = train_df[\"target\"].value_counts()\nval_counts","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:44:20.170213Z","iopub.execute_input":"2021-06-11T02:44:20.170566Z","iopub.status.idle":"2021-06-11T02:44:20.190210Z","shell.execute_reply.started":"2021-06-11T02:44:20.170534Z","shell.execute_reply":"2021-06-11T02:44:20.189221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có 1225312 câu hỏi chân thành và 80810 câu hỏi không chân thành","metadata":{}},{"cell_type":"code","source":"sincere_q = val_counts[0]/val_counts.sum()\nsincere_q = sincere_q*100\nprint('Có {}% câu hỏi là chân thành còn phần còn lại là không chân thành '.format(sincere_q))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:44:20.191503Z","iopub.execute_input":"2021-06-11T02:44:20.191812Z","iopub.status.idle":"2021-06-11T02:44:20.198668Z","shell.execute_reply.started":"2021-06-11T02:44:20.191782Z","shell.execute_reply":"2021-06-11T02:44:20.197770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Trực quan hóa dữ liệu**","metadata":{}},{"cell_type":"code","source":"trace = go.Bar(\n    x=val_counts.index,\n    y=val_counts.values,\n    marker=dict(\n        color=val_counts.values,\n        colorscale = 'Picnic',\n        reversescale = True\n    ),\n)\n\nlayout = go.Layout(\n    title='Lượng câu hỏi mỗi loại',\n    font=dict(size=14)\n)\n\ndata = [trace]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig, filename=\"TargetCount\")\n\nlabels = (np.array(val_counts.index))\nsizes = (np.array((val_counts / val_counts.sum())*100))\n\ntrace = go.Pie(labels=labels, values=sizes)\nlayout = go.Layout(\n    title='Phân phối loại câu hỏi',\n    font=dict(size=14),\n    width=500,\n    height=500,\n)\ndata = [trace]\nfig = go.Figure(data=data, layout=layout)\npy.iplot(fig, filename=\"usertype\")","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:44:20.201159Z","iopub.execute_input":"2021-06-11T02:44:20.201480Z","iopub.status.idle":"2021-06-11T02:44:21.326481Z","shell.execute_reply.started":"2021-06-11T02:44:20.201431Z","shell.execute_reply":"2021-06-11T02:44:21.325437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dựa vào tỉ lệ và biểu đồ ta thấy dữ liệu không cân bằng vì lượng câu hỏi chân thành lớn hơn nhiều so với lượng câu hỏi không chân thành.\n\n=> Sẽ so sánh kết quả model khi dữ liệu không cân bằng với model khi dữ liệu đã được resampling","metadata":{}},{"cell_type":"markdown","source":"### **Word Cloud**\n\nSử dụng thư viện WordCloud sẽ cho ta thấy tần suất xuất hiện các từ trong câu hỏi. Từ nào xuất hiện nhiều kích thước sẽ càng lớn và ngược lại.","metadata":{}},{"cell_type":"code","source":"from wordcloud import WordCloud","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:44:21.328291Z","iopub.execute_input":"2021-06-11T02:44:21.328594Z","iopub.status.idle":"2021-06-11T02:44:21.334699Z","shell.execute_reply.started":"2021-06-11T02:44:21.328562Z","shell.execute_reply":"2021-06-11T02:44:21.333788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere_wordcloud = WordCloud(width=5000, \n                                height=4000,\n                                colormap='Reds',\n                                background_color ='white', \n                                min_font_size = 8).generate(str(train_df[train_df[\"target\"] == 1][\"question_text\"]))\nplt.figure(figsize=(15,10), facecolor=None)\nplt.imshow(insincere_wordcloud)\nplt.axis(\"off\")\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:44:21.335929Z","iopub.execute_input":"2021-06-11T02:44:21.336242Z","iopub.status.idle":"2021-06-11T02:44:51.625411Z","shell.execute_reply.started":"2021-06-11T02:44:21.336210Z","shell.execute_reply":"2021-06-11T02:44:51.624297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Các từ có tần suất xuất hiện nhiều trong câu hỏi không chân thành","metadata":{}},{"cell_type":"code","source":"sincere_wordcloud = WordCloud(width=5000, \n                                height=4000,\n                                colormap='Greens',\n                                background_color ='white', \n                                min_font_size = 8).generate(str(train_df[train_df[\"target\"] == 0][\"question_text\"]))\nplt.figure(figsize=(15,10), facecolor=None)\nplt.imshow(sincere_wordcloud)\nplt.axis(\"off\")\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:44:51.626816Z","iopub.execute_input":"2021-06-11T02:44:51.627110Z","iopub.status.idle":"2021-06-11T02:45:16.822763Z","shell.execute_reply.started":"2021-06-11T02:44:51.627078Z","shell.execute_reply":"2021-06-11T02:45:16.821612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Các từ có tần suất xuất hiện nhiều trong câu hỏi chân thành\n","metadata":{}},{"cell_type":"markdown","source":"## **2. Tiền xử lý dữ liệu** ","metadata":{}},{"cell_type":"markdown","source":"**Tiền xử lý dữ liệu là một bước rất quan trọng trong việc giải quyết bất kỳ vấn đề nào trong lĩnh vực Học Máy. Hầu hết các bộ dữ liệu được sử dụng trong các vấn đề liên quan đến Học Máy cần được xử lý, làm sạch và biến đổi trước khi một mô hình có thể được huấn luyện trên những bộ dữ liệu này. Các kỹ thuật tiền xử lý dữ liệu được sử dụng trong bài:**\n* Xóa các stopwords: Stopwords là các từ có ít hoặc không có ý nghĩa gì đặc biệt khi xây dựng các đặc trưng. Đây thường là giới từ, trợ từ có tần suất xuất hiện tương đối cao trong một văn bản thông thường ví dụ như: a, an, the... Thư viện nltk có một danh sách các stopword có sẵn\n* Xử lý từ gốc và ngữ pháp: Trong các ngữ cảnh khác nhau, các từ gốc thường được gắn thêm các tiền tố và hậu tố vào để đúng với ngữ pháp. Ví dụ các từ: WATCHES, WATCHING, and WATCHED. Chúng ta có thể thấy rằng chúng đều có chung từ gốc là WATCH\n* Bên cạnh đó em cũng dùng tokenization, xóa bỏ các khoảng trắng thừa, chuẩn hóa chữ cái viết hoa, xử lý các kí tự số","metadata":{}},{"cell_type":"code","source":"import nltk\nimport string\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\n\nnltk.download('stopwords')\nnltk_stopwords = stopwords.words('english')\n\nwordnet_lemmatizer = WordNetLemmatizer()\n\ndef lemSentence(sentence):\n    token_words = word_tokenize(sentence) #Tokenize \n    lem_sentence = []\n    for word in token_words:\n        lem_sentence.append(wordnet_lemmatizer.lemmatize(word, pos=\"v\"))\n        lem_sentence.append(\" \")\n    return \"\".join(lem_sentence)\n\ndef clean(message, lem=True):\n    # Loại bỏ dấu câu\n    message = message.translate(str.maketrans('', '', string.punctuation))\n    \n    # Loại bỏ chữ số\n    message = message.translate(str.maketrans('', '', string.digits))\n    \n    # Loại bỏ stopwords\n    message = [word for word in word_tokenize(message) if not word.lower() in nltk_stopwords]\n    message = ' '.join(message)\n    \n    # Xử lý từ gốc và ngữ pháp\n    if lem:\n        message = lemSentence(message)\n    \n    return message","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:45:16.824136Z","iopub.execute_input":"2021-06-11T02:45:16.824444Z","iopub.status.idle":"2021-06-11T02:45:17.485960Z","shell.execute_reply.started":"2021-06-11T02:45:16.824413Z","shell.execute_reply":"2021-06-11T02:45:17.485008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['question_text_cleaned'] = train_df.question_text.apply(lambda x: clean(x, True))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:45:17.487351Z","iopub.execute_input":"2021-06-11T02:45:17.487637Z","iopub.status.idle":"2021-06-11T02:53:46.020422Z","shell.execute_reply.started":"2021-06-11T02:45:17.487609Z","shell.execute_reply":"2021-06-11T02:53:46.019278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['question_text_cleaned'] = test_df.question_text.apply(lambda x: clean(x, True))","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:53:46.021861Z","iopub.execute_input":"2021-06-11T02:53:46.022171Z","iopub.status.idle":"2021-06-11T02:56:11.892530Z","shell.execute_reply.started":"2021-06-11T02:53:46.022140Z","shell.execute_reply":"2021-06-11T02:56:11.891519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **3. Mô hình huấn luyện**","metadata":{}},{"cell_type":"markdown","source":"Chia tập dữ liệu thành 2 phần theo tỉ lệ 80-20:\n* Tập train dùng để huấn luyện model\n* Tập validation dùng để đánh giá model","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train, y_val = train_test_split(train_df['question_text_cleaned'], train_df.target, test_size=0.2, stratify = train_df.target.values)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:56:11.893891Z","iopub.execute_input":"2021-06-11T02:56:11.894224Z","iopub.status.idle":"2021-06-11T02:56:12.985251Z","shell.execute_reply.started":"2021-06-11T02:56:11.894191Z","shell.execute_reply":"2021-06-11T02:56:12.984224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import GridSearchCV, StratifiedKFold\n\n# #Setting the range for class weights\n# weights = np.linspace(0.0,0.99,200)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:56:12.988831Z","iopub.execute_input":"2021-06-11T02:56:12.989148Z","iopub.status.idle":"2021-06-11T02:56:12.992220Z","shell.execute_reply.started":"2021-06-11T02:56:12.989115Z","shell.execute_reply":"2021-06-11T02:56:12.991549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tạo dictionary cho grid search\n# param_grid = {'class_weight': [{0:x, 1:1.0-x} for x in weights]}","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:56:12.993423Z","iopub.execute_input":"2021-06-11T02:56:12.993882Z","iopub.status.idle":"2021-06-11T02:56:13.007911Z","shell.execute_reply.started":"2021-06-11T02:56:12.993837Z","shell.execute_reply":"2021-06-11T02:56:13.006634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting grid search vào train data với 5 folds\n# lr = LogisticRegression(solver='saga')\n\n# gridsearch = GridSearchCV(estimator= lr, \n#                           param_grid= param_grid,\n#                           cv=StratifiedKFold(), \n#                           n_jobs=-1, \n#                           scoring='f1', \n#                           verbose=2).fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:56:13.009354Z","iopub.execute_input":"2021-06-11T02:56:13.009795Z","iopub.status.idle":"2021-06-11T02:56:13.021069Z","shell.execute_reply.started":"2021-06-11T02:56:13.009760Z","shell.execute_reply":"2021-06-11T02:56:13.020023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Công cụ triển khai**\n**CountVectorizer**\n\nSau khi đã làm sạch dữ liệu văn bản, những từ này cần được mã hóa dưới dạng số để sử dụng vào trong thuật toán học máy. Quá trình này được gọi là trích xuất đặc trưng (vectơ hóa). CountVectorizer được sử dụng để chuyển đổi một bộ các tài liệu văn bản thành vectơ và đọc hiểu dữ liệu dựa vào tần số xuất hiện từ vựng đó.\n\nNhược điểm:\n* Không xác định được mức độ quan trọng giữa các từ.\n* Không xác định mối quan hệ giữa các từ với nhau (VD: sự tương đồng về mặt ngữ nghĩa)\n* Chỉ đánh giá những từ xuất hiện nhiều trong kho văn bản mới có ý nghĩa về mặt thống kê.\n\n**Logistic Regression**\n\nLogistic Regression là 1 thuật toán phân loại được dùng để gán các đối tượng cho 1 tập hợp giá trị rời rạc (như 0, 1, 2, ...). Một ví dụ điển hình là phân loại email, gồm có email công việc, email gia đình, email spam, ...\n=> Mô hình thích hợp cho bài toán\n\n**Pipeline**\n\n* Một pipeline trong sklearn là một tập các chuỗi thuật toán để trích xuất đặc trưng, tiền xử lý, chuyển hóa và huấn luyện dữ liệu sử dụng các thuật toán học máy cụ thể. \n* Mỗi pipeline bao gồm một vài bước nhất định, mỗi bước bao gồm một tham số là tên của bước, tham số còn lại là bộ chuyển đổi dữ liệu tương ứng (thường gọi là transformer). Bước cuối cùng trong một pipeline được gọi là estimator. Một estimator có thể là một thuật toán phân lớp, một thuật toán hồi quy, một mạng nơ-ron hay có thể là một thuật toán học máy không giám sát.\n* Để huấn luyện estimator tại bước cuối cùng của pipeline, ta phải gọi phương thức fit củapipeline và cung cấp dữ liệu để huấn luyện.\n* Một khi dữ liệu đã được huấn luyện bằng cách sử dụng estimator trong pipeline, ta có thể sử dụng pipeline đó để dự đoán đầu ra cho dữ liệu mới bằng cách sử dụng phương thức predict.\n\n\n\n","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.linear_model import LogisticRegression\n\npipeline_weight_cv = Pipeline([(\"cv\", CountVectorizer(analyzer=\"word\", ngram_range=(1,2), max_df=0.9)),\n                     (\"clf\", LogisticRegression(solver=\"saga\", class_weight={0: 0.0619, 1: 0.9381}, max_iter=10000, verbose=1, n_jobs=-1))])\n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:56:13.022962Z","iopub.execute_input":"2021-06-11T02:56:13.023352Z","iopub.status.idle":"2021-06-11T02:56:13.041867Z","shell.execute_reply.started":"2021-06-11T02:56:13.023297Z","shell.execute_reply":"2021-06-11T02:56:13.040654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Giá trị class_weight: Phạt các mẫu ở class[i] với class_weight[i]. Nghĩa là class_weight có giá trị càng lớn thì sẽ càng nhấn mạnh vào class đó.\n\nThay đổi class_weight: Vì sự mất cân bằng trong bộ dữ liệu (số lượng câu hỏi chân thành (0) quá nhiều (93.81%) so với số lượng câu hỏi không chân thành (1) (6.19%)) => Tăng giá trị class_weight vào 1 và giảm giá trị class_weight vào 0. ","metadata":{}},{"cell_type":"code","source":"def result(y_test, y_pred):\n    print(\"Accuracy: \", accuracy_score(y_test, y_pred))\n    print(\"F1-score: \",f1_score(y_test, y_pred, pos_label=1))\n    cm = confusion_matrix(y_test, y_pred)\n    sns.heatmap(cm/np.sum(cm), annot=True, \n            fmt='.2%', cmap='YlGnBu')","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:56:13.043291Z","iopub.execute_input":"2021-06-11T02:56:13.043642Z","iopub.status.idle":"2021-06-11T02:56:13.055606Z","shell.execute_reply.started":"2021-06-11T02:56:13.043586Z","shell.execute_reply":"2021-06-11T02:56:13.054402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_model_weight_cv = pipeline_weight_cv.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T02:56:13.057121Z","iopub.execute_input":"2021-06-11T02:56:13.057458Z","iopub.status.idle":"2021-06-11T03:13:27.818161Z","shell.execute_reply.started":"2021-06-11T02:56:13.057425Z","shell.execute_reply":"2021-06-11T03:13:27.817343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_weight_cv = lr_model_weight_cv.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:13:27.819212Z","iopub.execute_input":"2021-06-11T03:13:27.819626Z","iopub.status.idle":"2021-06-11T03:13:34.446555Z","shell.execute_reply.started":"2021-06-11T03:13:27.819595Z","shell.execute_reply":"2021-06-11T03:13:34.445457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result(y_val, y_pred_weight_cv)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:13:34.451130Z","iopub.execute_input":"2021-06-11T03:13:34.451462Z","iopub.status.idle":"2021-06-11T03:13:35.132457Z","shell.execute_reply.started":"2021-06-11T03:13:34.451427Z","shell.execute_reply":"2021-06-11T03:13:35.131464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Do sự mất cân bằng trong dữ liệu => Xảy ra hiện tượng overfitting khi có sự chênh lệch lớn giữa F1-score và Accuracy","metadata":{}},{"cell_type":"code","source":"pipeline_balanced_cv = Pipeline([(\"cv\", CountVectorizer(analyzer=\"word\", ngram_range=(1,2), max_df=0.9)),\n                     (\"clf\", LogisticRegression(solver=\"saga\", class_weight=\"balanced\", max_iter=10000, verbose=1, n_jobs=-1))])\n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:13:35.134473Z","iopub.execute_input":"2021-06-11T03:13:35.134786Z","iopub.status.idle":"2021-06-11T03:13:35.141494Z","shell.execute_reply.started":"2021-06-11T03:13:35.134754Z","shell.execute_reply":"2021-06-11T03:13:35.138812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Để giá trị class_weight = \"balanced\": Sử dụng tổng lượng mẫu để tự động điều chỉnh trọng số sao cho tỉ lệ nghịch với số mẫu của các lớp.","metadata":{}},{"cell_type":"code","source":"lr_model_balanced_cv = pipeline_balanced_cv.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:13:35.142774Z","iopub.execute_input":"2021-06-11T03:13:35.143064Z","iopub.status.idle":"2021-06-11T03:30:49.711513Z","shell.execute_reply.started":"2021-06-11T03:13:35.143034Z","shell.execute_reply":"2021-06-11T03:30:49.710258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_balanced_cv = lr_model_balanced_cv.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:30:49.712857Z","iopub.execute_input":"2021-06-11T03:30:49.713178Z","iopub.status.idle":"2021-06-11T03:30:56.187021Z","shell.execute_reply.started":"2021-06-11T03:30:49.713136Z","shell.execute_reply":"2021-06-11T03:30:56.186123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result(y_val, y_pred_balanced_cv)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:30:56.188200Z","iopub.execute_input":"2021-06-11T03:30:56.188662Z","iopub.status.idle":"2021-06-11T03:30:56.866803Z","shell.execute_reply.started":"2021-06-11T03:30:56.188630Z","shell.execute_reply":"2021-06-11T03:30:56.865718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Downsampling**\n\nDown sampling là việc giảm số lượng các mẫu của nhóm đa số để nó trở nên cân bằng với số mẫu của nhóm thiểu số. \n* Ưu điểm: Làm cân bằng mẫu một cách nhanh chóng, dễ dàng tiến hành thực hiện mà không cần đến thuật toán giả lập mẫu.\n* Nhược điểm: Kích thước mẫu sẽ bị giảm đáng kể.","metadata":{}},{"cell_type":"code","source":"insincere = train_df[train_df['target'] == 0]\nsincere = train_df[train_df['target'] == 1]","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:30:56.867948Z","iopub.execute_input":"2021-06-11T03:30:56.868219Z","iopub.status.idle":"2021-06-11T03:30:57.169407Z","shell.execute_reply.started":"2021-06-11T03:30:56.868191Z","shell.execute_reply":"2021-06-11T03:30:57.168328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Đảo câu hỏi chân thành = 1, câu hỏi không chân thành = 0 để tiến hành down sampling.","metadata":{}},{"cell_type":"code","source":"len(sincere)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:30:57.170770Z","iopub.execute_input":"2021-06-11T03:30:57.171066Z","iopub.status.idle":"2021-06-11T03:30:57.176939Z","shell.execute_reply.started":"2021-06-11T03:30:57.171036Z","shell.execute_reply":"2021-06-11T03:30:57.175933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lấy số lượng câu hỏi không chân thành, từ đó giảm số lượng câu hỏi chân thành xuống bằng với số lượng câu hỏi không chân thành.","metadata":{}},{"cell_type":"code","source":"insincere_batch = insincere[:len(sincere)]","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:30:57.178251Z","iopub.execute_input":"2021-06-11T03:30:57.178548Z","iopub.status.idle":"2021-06-11T03:30:57.191358Z","shell.execute_reply.started":"2021-06-11T03:30:57.178516Z","shell.execute_reply":"2021-06-11T03:30:57.190055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere_batch","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:30:57.192678Z","iopub.execute_input":"2021-06-11T03:30:57.193099Z","iopub.status.idle":"2021-06-11T03:30:57.219337Z","shell.execute_reply.started":"2021-06-11T03:30:57.193055Z","shell.execute_reply":"2021-06-11T03:30:57.218003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_batch = pd.concat([insincere_batch, sincere])","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:30:57.221130Z","iopub.execute_input":"2021-06-11T03:30:57.221571Z","iopub.status.idle":"2021-06-11T03:30:57.276434Z","shell.execute_reply.started":"2021-06-11T03:30:57.221523Z","shell.execute_reply":"2021-06-11T03:30:57.275434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ghép lượng câu hỏi chân thành và không chân thành vào với nhau.","metadata":{}},{"cell_type":"code","source":"test_batch","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:30:57.278008Z","iopub.execute_input":"2021-06-11T03:30:57.278408Z","iopub.status.idle":"2021-06-11T03:30:57.296040Z","shell.execute_reply.started":"2021-06-11T03:30:57.278362Z","shell.execute_reply":"2021-06-11T03:30:57.294966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train_down, X_val_down, y_train_down, y_val_down = train_test_split(test_batch['question_text_cleaned'], test_batch.target, test_size=0.2, stratify = test_batch.target.values)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:30:57.297557Z","iopub.execute_input":"2021-06-11T03:30:57.297990Z","iopub.status.idle":"2021-06-11T03:30:57.486315Z","shell.execute_reply.started":"2021-06-11T03:30:57.297945Z","shell.execute_reply":"2021-06-11T03:30:57.485350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Fine tuning giá trị C**\n\nGiá trị C: Chính quy hóa (Regularization): Điều chỉnh số lỗi bằng cách sử dụng hàm hợp lý trên tập training và tránh overfitting trong khi vẫn giữ được tính tổng quát của nó.\n\nThay đổi các giá trị C để tìm được model phù hợp.","metadata":{}},{"cell_type":"code","source":"C_param_range = np.arange(0.1, 0.501, 0.05)\n\ntrain_acc_table_cv = pd.DataFrame(columns = ['C_parameter','Accuracy', 'F1-Score'])\ntrain_acc_table_cv['C_parameter'] = C_param_range\n\nj = 0\nfor i in C_param_range:\n    \n    pipeline_down_C_cv = Pipeline([(\"cv\", CountVectorizer(analyzer=\"word\", ngram_range=(1,2), max_df=0.9)),\n                     (\"clf\", LogisticRegression(solver=\"saga\", class_weight=\"balanced\", C=i, max_iter=10000, verbose=1, n_jobs=-1))])\n\n    lr_down_C_cv = pipeline_down_C_cv.fit(X_train_down, y_train_down)\n    \n    y_pred_down_C_cv = lr_down_C_cv.predict(X_val_down)\n    \n    train_acc_table_cv.iloc[j,1] = accuracy_score(y_val_down, y_pred_down_C_cv)\n    train_acc_table_cv.iloc[j,2] = f1_score(y_val_down, y_pred_down_C_cv)\n\n    j += 1\n       ","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:30:57.487872Z","iopub.execute_input":"2021-06-11T03:30:57.488259Z","iopub.status.idle":"2021-06-11T03:34:17.347123Z","shell.execute_reply.started":"2021-06-11T03:30:57.488215Z","shell.execute_reply":"2021-06-11T03:34:17.346014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_acc_table_cv","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:34:17.348431Z","iopub.execute_input":"2021-06-11T03:34:17.348769Z","iopub.status.idle":"2021-06-11T03:34:17.361272Z","shell.execute_reply.started":"2021-06-11T03:34:17.348734Z","shell.execute_reply":"2021-06-11T03:34:17.360168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Chọn giá trị C = 0.45 có kết quả Accuracy và F1-score tốt nhất","metadata":{}},{"cell_type":"code","source":"pipeline_down_best_cv = Pipeline([(\"cv\", CountVectorizer(analyzer=\"word\", ngram_range=(1,2), max_df=0.9)),\n                     (\"clf\", LogisticRegression(solver=\"saga\", class_weight=\"balanced\", C=0.45, max_iter=10000, verbose=1, n_jobs=-1))])\n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:35:04.149775Z","iopub.execute_input":"2021-06-11T03:35:04.150144Z","iopub.status.idle":"2021-06-11T03:35:04.156296Z","shell.execute_reply.started":"2021-06-11T03:35:04.150114Z","shell.execute_reply":"2021-06-11T03:35:04.154838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_down_best_cv = pipeline_down_best_cv.fit(X_train_down, y_train_down)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_down_best_cv = lr_down_best_cv.predict(X_val_down)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result(y_val_down, y_pred_down_best_cv)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TF-IDF vectorize**\n\nTF-IDF là viết tắt từ cụm từ tiếng Anh: term frequency–inverse document frequency, là một thống kê số học nhằm phản ánh tầm quan trọng của một từ đối với một văn bản trong một tập hợp hay một ngữ liệu văn bản. TF–IDF thường dùng dưới dạng là một trọng số trong tìm kiếm truy xuất thông tin, khai thác văn bản, và mô hình hóa người dùng.\n\nTF-IDF được đánh giá là tốt hơn CountVectorizer vì ngoài tập trung vào tần số xuất hiện của một từ, nó còn cho ra độ quan trọng của từ đó trong kho văn bản => Loại bỏ những từ ít quan trọng, model sẽ bớt phức tạp","metadata":{}},{"cell_type":"markdown","source":"Weight","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\n\npipeline_weight_tfidf = Pipeline([(\"tfidf\", TfidfVectorizer(analyzer=\"word\", ngram_range=(1,2), max_df=0.9)),\n                     (\"clf\", LogisticRegression(solver=\"saga\", class_weight={0: 0.0619, 1: 0.9381}, max_iter=10000, verbose=1, n_jobs=-1))])\n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:35:53.727091Z","iopub.execute_input":"2021-06-11T03:35:53.727468Z","iopub.status.idle":"2021-06-11T03:35:53.733088Z","shell.execute_reply.started":"2021-06-11T03:35:53.727433Z","shell.execute_reply":"2021-06-11T03:35:53.732108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_model_weight_tfidf = pipeline_weight_tfidf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:35:56.285383Z","iopub.execute_input":"2021-06-11T03:35:56.286035Z","iopub.status.idle":"2021-06-11T03:37:07.984647Z","shell.execute_reply.started":"2021-06-11T03:35:56.285984Z","shell.execute_reply":"2021-06-11T03:37:07.983642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_weight_tfidf = lr_model_weight_tfidf.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:37:07.986444Z","iopub.execute_input":"2021-06-11T03:37:07.986936Z","iopub.status.idle":"2021-06-11T03:37:14.586552Z","shell.execute_reply.started":"2021-06-11T03:37:07.986887Z","shell.execute_reply":"2021-06-11T03:37:14.585462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result(y_val, y_pred_weight_tfidf)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:38:30.389628Z","iopub.execute_input":"2021-06-11T03:38:30.390024Z","iopub.status.idle":"2021-06-11T03:38:31.063532Z","shell.execute_reply.started":"2021-06-11T03:38:30.389990Z","shell.execute_reply":"2021-06-11T03:38:31.062331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Balanced","metadata":{}},{"cell_type":"code","source":"pipeline_balanced_tfidf = Pipeline([(\"tfidf\", TfidfVectorizer(analyzer=\"word\", ngram_range=(1,2), max_df=0.9)),\n                     (\"clf\", LogisticRegression(solver=\"saga\", class_weight=\"balanced\", max_iter=10000, verbose=1, n_jobs=-1))])\n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T03:38:47.265760Z","iopub.execute_input":"2021-06-11T03:38:47.266112Z","iopub.status.idle":"2021-06-11T03:38:47.271007Z","shell.execute_reply.started":"2021-06-11T03:38:47.266080Z","shell.execute_reply":"2021-06-11T03:38:47.270232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lr_model_balanced_tfidf = pipeline_balanced_tfidf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:20:23.543386Z","iopub.execute_input":"2021-06-11T04:20:23.543977Z","iopub.status.idle":"2021-06-11T04:20:24.295736Z","shell.execute_reply.started":"2021-06-11T04:20:23.543931Z","shell.execute_reply":"2021-06-11T04:20:24.294048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred_balanced_tfidf = lr_model_balanced_tfidf.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:20:22.568474Z","iopub.status.idle":"2021-06-11T04:20:22.569459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# result(y_val, y_pred_balanced_tfidf)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:20:22.570958Z","iopub.status.idle":"2021-06-11T04:20:22.571886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fine tuning C","metadata":{}},{"cell_type":"code","source":"C_param_range = np.arange(0.1, 0.501, 0.05)\n\ntrain_acc_table_tfidf = pd.DataFrame(columns = ['C_parameter','Accuracy', 'F1-Score'])\ntrain_acc_table_tfidf['C_parameter'] = C_param_range\n\nj = 0\nfor i in C_param_range:\n    \n    pipeline_down_C_tfidf = Pipeline([(\"cv\", CountVectorizer(analyzer=\"word\", ngram_range=(1,2), max_df=0.9)),\n                     (\"clf\", LogisticRegression(solver=\"saga\", class_weight=\"balanced\", C=i, max_iter=10000, verbose=1, n_jobs=-1))])\n\n    lr_down_C_tfidf = pipeline_down_C_tfidf.fit(X_train_down, y_train_down)\n    \n    y_pred_down_C_tfidf = lr_down_C_tfidf.predict(X_val_down)\n    \n    train_acc_table_tfidf.iloc[j,1] = accuracy_score(y_val_down, y_pred_down_C_tfidf)\n    train_acc_table_tfidf.iloc[j,2] = f1_score(y_val_down, y_pred_down_C_tfidf)\n\n    j += 1","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:20:31.443314Z","iopub.execute_input":"2021-06-11T04:20:31.443916Z","iopub.status.idle":"2021-06-11T04:23:55.155696Z","shell.execute_reply.started":"2021-06-11T04:20:31.443877Z","shell.execute_reply":"2021-06-11T04:23:55.154880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_acc_table_tfidf","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:23:55.157081Z","iopub.execute_input":"2021-06-11T04:23:55.157558Z","iopub.status.idle":"2021-06-11T04:23:55.170885Z","shell.execute_reply.started":"2021-06-11T04:23:55.157525Z","shell.execute_reply":"2021-06-11T04:23:55.170070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Chọn C = 0.5 cho ra kết quả Accuracy và F1-score tốt nhất.","metadata":{}},{"cell_type":"code","source":"pipeline_down_best_tfidf = Pipeline([(\"cv\", CountVectorizer(analyzer=\"word\", ngram_range=(1,2), max_df=0.9)),\n                     (\"clf\", LogisticRegression(solver=\"saga\", class_weight=\"balanced\", C=0.5, max_iter=10000, verbose=1, n_jobs=-1))])\n","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:25:01.383931Z","iopub.execute_input":"2021-06-11T04:25:01.384291Z","iopub.status.idle":"2021-06-11T04:25:01.390295Z","shell.execute_reply.started":"2021-06-11T04:25:01.384261Z","shell.execute_reply":"2021-06-11T04:25:01.389113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_down_best_tfidf = pipeline_down_best_tfidf.fit(X_train_down, y_train_down)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:25:18.960195Z","iopub.execute_input":"2021-06-11T04:25:18.960592Z","iopub.status.idle":"2021-06-11T04:25:49.917076Z","shell.execute_reply.started":"2021-06-11T04:25:18.960551Z","shell.execute_reply":"2021-06-11T04:25:49.915985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred_down_best_tfidf = lr_down_best_tfidf.predict(X_val_down)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:25:49.918603Z","iopub.execute_input":"2021-06-11T04:25:49.919029Z","iopub.status.idle":"2021-06-11T04:25:50.834967Z","shell.execute_reply.started":"2021-06-11T04:25:49.918985Z","shell.execute_reply":"2021-06-11T04:25:50.833839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result(y_val_down, y_pred_down_best_tfidf)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:28:56.481594Z","iopub.execute_input":"2021-06-11T04:28:56.481995Z","iopub.status.idle":"2021-06-11T04:28:57.015273Z","shell.execute_reply.started":"2021-06-11T04:28:56.481958Z","shell.execute_reply":"2021-06-11T04:28:57.014493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **4. Submission**","metadata":{}},{"cell_type":"code","source":"test_df['prediction'] = lr_down_best_tfidf.predict(test_df['question_text_cleaned'])","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:29:03.518225Z","iopub.execute_input":"2021-06-11T04:29:03.518927Z","iopub.status.idle":"2021-06-11T04:29:12.132136Z","shell.execute_reply.started":"2021-06-11T04:29:03.518877Z","shell.execute_reply":"2021-06-11T04:29:12.131052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = test_df[['qid','prediction']]\nresults.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:29:23.404600Z","iopub.execute_input":"2021-06-11T04:29:23.405206Z","iopub.status.idle":"2021-06-11T04:29:24.351835Z","shell.execute_reply.started":"2021-06-11T04:29:23.405149Z","shell.execute_reply":"2021-06-11T04:29:24.350632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-06-11T04:29:31.727761Z","iopub.execute_input":"2021-06-11T04:29:31.728174Z","iopub.status.idle":"2021-06-11T04:29:31.739267Z","shell.execute_reply.started":"2021-06-11T04:29:31.728129Z","shell.execute_reply":"2021-06-11T04:29:31.738084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Hướng cải thiện**\n* Tìm cách fine tuning model\n* Sử dụng model khác","metadata":{}}]}