{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-04T03:33:34.799949Z","iopub.execute_input":"2021-06-04T03:33:34.800351Z","iopub.status.idle":"2021-06-04T03:33:34.805909Z","shell.execute_reply.started":"2021-06-04T03:33:34.800266Z","shell.execute_reply":"2021-06-04T03:33:34.804823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # **Import packages and libraries**","metadata":{}},{"cell_type":"code","source":"import re\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt \nimport seaborn as sns\nimport nltk\nimport pickle\n\nfrom nltk import word_tokenize\nfrom nltk.corpus import stopwords\nfrom nltk.stem.wordnet import WordNetLemmatizer\nfrom nltk.tokenize import word_tokenize\nfrom collections import Counter\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import CountVectorizer","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:13:11.939814Z","iopub.execute_input":"2021-06-10T13:13:11.940313Z","iopub.status.idle":"2021-06-10T13:13:13.841676Z","shell.execute_reply.started":"2021-06-10T13:13:11.940281Z","shell.execute_reply":"2021-06-10T13:13:13.840893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # **EDA**","metadata":{}},{"cell_type":"markdown","source":"### Đọc file dữ liệu","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ntest = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:13:13.842955Z","iopub.execute_input":"2021-06-10T13:13:13.843383Z","iopub.status.idle":"2021-06-10T13:13:19.977653Z","shell.execute_reply.started":"2021-06-10T13:13:13.843354Z","shell.execute_reply":"2021-06-10T13:13:19.976782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Đầu tiên, hãy nhìn qua về tập dữ liệu train trước","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:13:19.979571Z","iopub.execute_input":"2021-06-10T13:13:19.980227Z","iopub.status.idle":"2021-06-10T13:13:20.009682Z","shell.execute_reply.started":"2021-06-10T13:13:19.980183Z","shell.execute_reply":"2021-06-10T13:13:20.008387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:13:20.011383Z","iopub.execute_input":"2021-06-10T13:13:20.011761Z","iopub.status.idle":"2021-06-10T13:13:20.272246Z","shell.execute_reply.started":"2021-06-10T13:13:20.011722Z","shell.execute_reply":"2021-06-10T13:13:20.271192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Bộ dữ liệu có **3 cột** và **1306122 non-null bản ghi (hàng)**   \nCột thứ 1 là '**qid**': id của từng câu hỏi  \nCột thứ 2 là '**question_text**': chứa tất cả các câu hỏi   \nCột thứ 3 là '**target**': nhãn của từng câu hỏi","metadata":{}},{"cell_type":"markdown","source":"Đếm số nhãn và số lượng của từng nhãn","metadata":{}},{"cell_type":"code","source":"labels = Counter(train['target']).keys()\namounts = Counter(train['target']).values()\nlabels, amounts","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:13:20.273692Z","iopub.execute_input":"2021-06-10T13:13:20.273990Z","iopub.status.idle":"2021-06-10T13:13:20.727127Z","shell.execute_reply.started":"2021-06-10T13:13:20.273960Z","shell.execute_reply":"2021-06-10T13:13:20.726292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Có 2 nhãn là: [0, 1]  với số lượng lần lượt là:   \n| **Nhãn '0'** | **Nhãn '1'** |  \n| :-------: | --------: |  \n| **1225312**   | **80810**     |","metadata":{}},{"cell_type":"markdown","source":"### Visualize data","metadata":{}},{"cell_type":"code","source":"fig = plt.figure()\nax = fig.add_axes([0,0,1,1])\n_labels = ['0', '1']\ncolors_list = ['#5cb85c','#5bc0de']\nax.bar(_labels, amounts, color=colors_list)\n\nplt.xlabel('Label')\nplt.ylabel('Amount')\n\nfor i in ax.patches:\n    height = i.get_height()\n    width = i.get_width()\n    x, y = i.get_xy()\n    ax.annotate(f'{height/1306122:.0%}', (x + width/2, height), ha='center')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:13:27.577881Z","iopub.execute_input":"2021-06-10T13:13:27.578473Z","iopub.status.idle":"2021-06-10T13:13:27.740946Z","shell.execute_reply.started":"2021-06-10T13:13:27.578437Z","shell.execute_reply":"2021-06-10T13:13:27.740114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sincere_percent= (len(train.question_text[train['target'] == 0]) /  len(train['question_text']) * 100)\ninsincere_percent= (len(train.question_text[train['target'] == 1]) / len(train['question_text']) * 100)\nsincere_percent, insincere_percent","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:13:46.915029Z","iopub.execute_input":"2021-06-10T13:13:46.915590Z","iopub.status.idle":"2021-06-10T13:13:47.003157Z","shell.execute_reply.started":"2021-06-10T13:13:46.915556Z","shell.execute_reply":"2021-06-10T13:13:47.001819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"__labels = 'Sincere', 'Insincere'\nsizes = [sincere_percent, insincere_percent]\nexplode = (0.1, 0)  # explode 1st slice\n\nplt.pie(sizes, explode=explode, labels=__labels, colors=colors_list,\nautopct='%1.1f%%', shadow=True, startangle=140)\n\nplt.axis('equal')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:13:47.872196Z","iopub.execute_input":"2021-06-10T13:13:47.872561Z","iopub.status.idle":"2021-06-10T13:13:48.008342Z","shell.execute_reply.started":"2021-06-10T13:13:47.872530Z","shell.execute_reply":"2021-06-10T13:13:48.007204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy là dữ liệu train có nhãn rất lệch:  \nNhãn dương '1' chỉ chiếm 6,2% tổng số nhãn.  \nĐiều này sẽ là một vấn đề lớn cần được giải quyết để mô hình có thể trở thành 1 classifier tốt.","metadata":{}},{"cell_type":"markdown","source":"### Analyzing text statistics","metadata":{}},{"cell_type":"markdown","source":"Biểu đồ số ký tự có trong từng câu","metadata":{}},{"cell_type":"code","source":"train['question_text'].str.len().hist()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:13:53.239986Z","iopub.execute_input":"2021-06-10T13:13:53.240364Z","iopub.status.idle":"2021-06-10T13:13:54.417202Z","shell.execute_reply.started":"2021-06-10T13:13:53.240331Z","shell.execute_reply":"2021-06-10T13:13:54.416234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy là các câu có độ dài thường từ 1->100 kí tự","metadata":{}},{"cell_type":"markdown","source":"Histogram về Số lượng từ trong 1 câu ","metadata":{}},{"cell_type":"code","source":"train['question_text'].str.split().map(lambda x: len(x)).hist()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:13:54.868084Z","iopub.execute_input":"2021-06-10T13:13:54.868447Z","iopub.status.idle":"2021-06-10T13:14:00.982227Z","shell.execute_reply.started":"2021-06-10T13:13:54.868417Z","shell.execute_reply":"2021-06-10T13:14:00.981201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Và 1 câu thường có 1->12 từ, ngoài ra, các câu có 13->30 từ cũng có số lượng tương đối","metadata":{}},{"cell_type":"markdown","source":"Tạo ra 1 corpus  \nCorpus: 1 list chứa tất cả các từ có trong tất cả các câu hỏi ở train['question_text']","metadata":{}},{"cell_type":"code","source":"corpus = []\nquestion = train['question_text'].str.split()\nquestion = question.values.tolist()\ncorpus = [word for q in question for word in q]","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:14:08.348666Z","iopub.execute_input":"2021-06-10T13:14:08.349026Z","iopub.status.idle":"2021-06-10T13:14:13.419052Z","shell.execute_reply.started":"2021-06-10T13:14:08.348995Z","shell.execute_reply":"2021-06-10T13:14:13.418037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(corpus), corpus[:20]","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:15:25.292312Z","iopub.execute_input":"2021-06-10T13:15:25.292837Z","iopub.status.idle":"2021-06-10T13:15:25.300032Z","shell.execute_reply.started":"2021-06-10T13:15:25.292806Z","shell.execute_reply":"2021-06-10T13:15:25.299191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Stopword thường là các từ được sử dụng phổ biến trong văn bản cũng như là trong giao tiếp hàng ngày. Các từ stopword tiếng anh thường là 'a', 'an', 'the',... Tuy nhiên, các từ này thường không mang nhiều ý nghĩa trong việc xác định đặc tính của câu, mà chúng thường được sử dụng rất nhiều cho nên sẽ gây ảnh hưởng tới bước feature-extraction.  \nTa sẽ in ra các stopword được sử dụng ở trong tập dữ liệu train để đánh giá về số lượng của chúng","metadata":{}},{"cell_type":"markdown","source":"Trước tiên, vì các câu hỏi đều là câu hỏi bằng tiếng anh nên tập stopwords em sử dụng sẽ là list stopwords với ngôn ngữ là 'english' từ thư viện nltk (thư viện Natural Language Tool Kit - 1 thư viện Python phổ biến được sử dụng trong các bài toán NLP) ","metadata":{}},{"cell_type":"code","source":"stop_words = stopwords.words('english')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:15:40.205745Z","iopub.execute_input":"2021-06-10T13:15:40.206352Z","iopub.status.idle":"2021-06-10T13:15:40.227608Z","shell.execute_reply.started":"2021-06-10T13:15:40.206298Z","shell.execute_reply":"2021-06-10T13:15:40.226419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(stop_words), stop_words)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:15:41.939839Z","iopub.execute_input":"2021-06-10T13:15:41.940252Z","iopub.status.idle":"2021-06-10T13:15:41.945398Z","shell.execute_reply.started":"2021-06-10T13:15:41.940218Z","shell.execute_reply":"2021-06-10T13:15:41.944612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# stopword_dict -> sw_dict\nfrom collections import defaultdict\nsw_dict = defaultdict(int)\nfor word in corpus:\n    if word in stop_words:\n        sw_dict[word]+=1","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:15:43.482451Z","iopub.execute_input":"2021-06-10T13:15:43.483008Z","iopub.status.idle":"2021-06-10T13:16:11.295657Z","shell.execute_reply.started":"2021-06-10T13:15:43.482953Z","shell.execute_reply":"2021-06-10T13:16:11.294605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(sw_dict), sw_dict)\nprint('The amount of stopwords: ', sum(sw_dict.values()))\nprint('Total words in corpus:', len(corpus))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:16:11.297060Z","iopub.execute_input":"2021-06-10T13:16:11.297427Z","iopub.status.idle":"2021-06-10T13:16:11.303529Z","shell.execute_reply.started":"2021-06-10T13:16:11.297393Z","shell.execute_reply":"2021-06-10T13:16:11.302425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có 167 trên 179 stopword được dùng và số lượng stopword này là vô cùng nhiều: Chiếm 38% số lượng từ.  \nVì vậy, trong phần Tiền xử lý tiếp theo, chúng ta sẽ loại bỏ đi tất cả các stopword này.","metadata":{}},{"cell_type":"markdown","source":"Một vài biểu đồ trực quan hóa","metadata":{}},{"cell_type":"code","source":"def plot_top_stopwords_barchart(texts):\n    ques = texts.str.split()\n    ques = ques.values.tolist()\n    corpus = [word for q in ques for word in q]\n    from collections import defaultdict\n    dic = defaultdict(int)\n    for word in corpus:\n        if word in stop_words:\n            dic[word]+=1\n            \n    top = sorted(dic.items(), key=lambda x:x[1],reverse=True)[:10] \n    x,y = zip(*top)\n    plt.bar(x,y)\n    plt.title('Top 10 stopword xuất hiện nhiều nhất')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:16:11.305759Z","iopub.execute_input":"2021-06-10T13:16:11.306251Z","iopub.status.idle":"2021-06-10T13:16:11.579778Z","shell.execute_reply.started":"2021-06-10T13:16:11.306205Z","shell.execute_reply":"2021-06-10T13:16:11.578771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_top_stopwords_barchart(train['question_text'])","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:16:24.947778Z","iopub.execute_input":"2021-06-10T13:16:24.948323Z","iopub.status.idle":"2021-06-10T13:16:59.262954Z","shell.execute_reply.started":"2021-06-10T13:16:24.948290Z","shell.execute_reply":"2021-06-10T13:16:59.262027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Vì đặc trưng của 1 câu trong văn bản có thể không chỉ được thể hiện qua 1 từ mà là còn từ cụm 2, 3 từ cho nên chúng ta sẽ:  \nVisualize các từ đơn, từ đôi, cụm 3 từ xuất hiện nhiều trong dữ liệu:","metadata":{}},{"cell_type":"code","source":"# Sử dụng hàm Counter từ thư viện collections để thống kê số lượng các từ đơn giản hơn\ncounter = Counter(corpus)\nmost = counter.most_common()\n\nx,y= [],[]\nfor word,count in most[:40]:\n    if (word not in stop_words):\n        x.append(word)\n        y.append(count)\n        \nsns.barplot(x=y,y=x).set(title='Các từ khác (không phải stopword) xuất hiện nhiều trong dữ liệu')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:16:59.264353Z","iopub.execute_input":"2021-06-10T13:16:59.264648Z","iopub.status.idle":"2021-06-10T13:17:03.486126Z","shell.execute_reply.started":"2021-06-10T13:16:59.264617Z","shell.execute_reply":"2021-06-10T13:17:03.485155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hàm tìm các từ đôi trong từng câu hỏi từ bộ dữ liệu\ndef get_top_ngram(corpus, n=None):\n    vec = CountVectorizer(ngram_range=(n, n)).fit(corpus)\n    bag_of_words = vec.transform(corpus)\n    sum_words = bag_of_words.sum(axis=0) \n    words_freq = [(word, sum_words[0, idx]) \n                  for word, idx in vec.vocabulary_.items()]\n    words_freq =sorted(words_freq, key = lambda x: x[1], reverse=True)\n    return words_freq[:10]","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:17:03.489893Z","iopub.execute_input":"2021-06-10T13:17:03.490248Z","iopub.status.idle":"2021-06-10T13:17:03.497273Z","shell.execute_reply.started":"2021-06-10T13:17:03.490216Z","shell.execute_reply":"2021-06-10T13:17:03.495955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Top từ đôi xuất hiện nhiều trong tập dữ liệu câu hỏi\ntop_bi_grams = get_top_ngram(train['question_text'],n=2)\nx,y = map(list,zip(*top_bi_grams)) \nsns.barplot(x=y,y=x).set(title='Các từ đôi xuất hiện nhiều nhất trong dữ liệu')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:17:03.498846Z","iopub.execute_input":"2021-06-10T13:17:03.499292Z","iopub.status.idle":"2021-06-10T13:18:45.025438Z","shell.execute_reply.started":"2021-06-10T13:17:03.499247Z","shell.execute_reply":"2021-06-10T13:18:45.024436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Top cụm 3 từ xuất hiện nhiều trong tập dữ liệu câu hỏi\ntop_tri_grams = get_top_ngram(train['question_text'],n=3)\nx,y = map(list,zip(*top_tri_grams)) \nsns.barplot(x=y,y=x).set(title='Các cụm 3 từ xuất hiện nhiều nhất trong dữ liệu')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:18:45.026698Z","iopub.execute_input":"2021-06-10T13:18:45.027017Z","iopub.status.idle":"2021-06-10T13:21:13.203216Z","shell.execute_reply.started":"2021-06-10T13:18:45.026987Z","shell.execute_reply":"2021-06-10T13:21:13.202252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thế thấy, khi chưa tiền xử lý dữ liệu, các từ đơn, từ đôi, cụm 3 từ xuất hiện nhiều nhất thường là các từ mở đầu câu hỏi như 'What', 'How', 'what is', 'is the', 'what is the', 'what are the', ... Điều này là không khó hiểu vì bộ dữ liệu là về các câu hỏi thu thập trên Quora.  \nTuy nhiên, nếu để những từ này lại thì sẽ gây cản trở cho việc học của mô hình vì đây là những từ không mang nhiều ý nghĩa nhưng xuất hiện nhiều (giống như stopword).  \nVậy nên, em sẽ đi vào bước tiếp theo là bước Tiền xử lý","metadata":{}},{"cell_type":"markdown","source":"-------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"Em sẽ thực hiện các bước **Tiền xử lý** cơ bản của bài toán NLP như:  \n* Lower các từ\n* Xóa các kí tự không phải là chữ cái\n* Xóa bỏ stopword\n* Lemmatation","metadata":{}},{"cell_type":"markdown","source":"> # **Tiền xử lý** ","metadata":{}},{"cell_type":"code","source":"# Xóa bỏ các từ là stopword\ndef delStopwords(text):\n    return ' '.join([w for w in word_tokenize(text) if not w in stop_words])\n\n# Đưa các từ về dạng từ gốc của nó\nwordnet = WordNetLemmatizer() \ndef lemmatizeText(text):\n    return ' '.join([wordnet.lemmatize(w, pos='v') for w in word_tokenize(text)])","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:23:54.470095Z","iopub.execute_input":"2021-06-10T13:23:54.470468Z","iopub.status.idle":"2021-06-10T13:23:54.476567Z","shell.execute_reply.started":"2021-06-10T13:23:54.470436Z","shell.execute_reply":"2021-06-10T13:23:54.475428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_text(text):\n    cleaned_text = re.sub('[\\n]',' ',text)    #xóa các ký tự xuống dòng\n    cleaned_text = re.sub('[A-Z]+', lambda m: m.group(0).lower(), cleaned_text)    #lower các ký tự \n    cleaned_text = re.sub('[^a-zA-Z]',' ',cleaned_text).strip()    #xóa các ký tự không phải là chữ cái\n    cleaned_text = delStopwords(cleaned_text)    #xóa stopwords\n    cleaned_text = lemmatizeText(cleaned_text)    #lemmatize các từ về danh noun\n    return cleaned_text","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:24:01.764819Z","iopub.execute_input":"2021-06-10T13:24:01.765201Z","iopub.status.idle":"2021-06-10T13:24:01.770873Z","shell.execute_reply.started":"2021-06-10T13:24:01.765168Z","shell.execute_reply":"2021-06-10T13:24:01.769719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thực hiện tiền xử lý dữ liệu và chia cả dữ liệu thành tập câu hỏi và tập nhãn","metadata":{}},{"cell_type":"code","source":"train_text = []\ntarget = []\nfor i in range(len(train)):\n    train_text.append(clean_text(train['question_text'][i]))\n    target.append((train['target'][i]).astype('int32'))\n    \ntest_text = []\nfor i in range(len(test)):\n    test_text.append(clean_text(test['question_text'][i]))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:25:32.373178Z","iopub.execute_input":"2021-06-10T13:25:32.373673Z","iopub.status.idle":"2021-06-10T13:37:37.275857Z","shell.execute_reply.started":"2021-06-10T13:25:32.373641Z","shell.execute_reply":"2021-06-10T13:37:37.274651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_text), len(target), len(test_text)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:37:37.277296Z","iopub.execute_input":"2021-06-10T13:37:37.277558Z","iopub.status.idle":"2021-06-10T13:37:37.285175Z","shell.execute_reply.started":"2021-06-10T13:37:37.277532Z","shell.execute_reply":"2021-06-10T13:37:37.283959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with open('./train_text.pkl', 'wb') as train_text_file:\n#     pickle.dump(train_text, train_text_file)\n# with open('./target.pkl', 'wb') as target_file:\n#     pickle.dump(target, target_file)\n# with open('./test_text.pkl', 'wb') as test_text_file:\n#     pickle.dump(test_text, test_text_file)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:37:53.075857Z","iopub.execute_input":"2021-06-10T13:37:53.076246Z","iopub.status.idle":"2021-06-10T13:37:53.080566Z","shell.execute_reply.started":"2021-06-10T13:37:53.076210Z","shell.execute_reply":"2021-06-10T13:37:53.079368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_text = pd.read_pickle(r'../input/preprocessedtexts/texts.pkl')\n# target = pd.read_pickle(r'../input/text-targets/targets.pkl')\n# test_text = pd.read_pickle(r'../input/preprocessedtesttexts/test_texts.pkl')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:37:53.674065Z","iopub.execute_input":"2021-06-10T13:37:53.674431Z","iopub.status.idle":"2021-06-10T13:37:53.677509Z","shell.execute_reply.started":"2021-06-10T13:37:53.674400Z","shell.execute_reply":"2021-06-10T13:37:53.676777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Như vậy, ta đã có được tập dữ liệu sau khi **Tiền xử lý**  ","metadata":{}},{"cell_type":"code","source":"preprocessed_train = pd.DataFrame(train_text, columns=['preprocessed_train_text'])\npreprocessed_test = pd.DataFrame(test_text, columns=['preprocessed_test_text'])","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:37:56.008197Z","iopub.execute_input":"2021-06-10T13:37:56.008713Z","iopub.status.idle":"2021-06-10T13:37:56.160999Z","shell.execute_reply.started":"2021-06-10T13:37:56.008679Z","shell.execute_reply":"2021-06-10T13:37:56.160190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Xây dựng 1 corpus mới với dữ liệu đã Tiền xử lý","metadata":{}},{"cell_type":"code","source":"new_corpus = []\nquest = preprocessed_train['preprocessed_train_text'].str.split()\nquest = quest.values.tolist()\nnew_corpus = [word for q in quest for word in q]","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:37:59.485544Z","iopub.execute_input":"2021-06-10T13:37:59.486066Z","iopub.status.idle":"2021-06-10T13:38:04.235582Z","shell.execute_reply.started":"2021-06-10T13:37:59.486032Z","shell.execute_reply":"2021-06-10T13:38:04.234674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counter = Counter(new_corpus)\nmost = counter.most_common()\n\nx,y= [],[]\nfor word,count in most[:20]:\n    x.append(word)\n    y.append(count)\n\nsns.barplot(x=y,y=x).set(title='Các từ xuất hiện nhiều trong dữ liệu đã Tiền xử lý')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:38:04.236838Z","iopub.execute_input":"2021-06-10T13:38:04.237291Z","iopub.status.idle":"2021-06-10T13:38:06.723150Z","shell.execute_reply.started":"2021-06-10T13:38:04.237258Z","shell.execute_reply":"2021-06-10T13:38:06.722410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Top từ đôi xuất hiện nhiều trong tập dữ liệu đã Tiền xử lý\ntop_bi_grams = get_top_ngram(preprocessed_train['preprocessed_train_text'],n=2)\nx,y = map(list,zip(*top_bi_grams)) \nsns.barplot(x=y,y=x).set(title='Các từ đôi xuất hiện nhiều nhất trong dữ liệu đã tiền xử lý')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:38:06.724427Z","iopub.execute_input":"2021-06-10T13:38:06.724799Z","iopub.status.idle":"2021-06-10T13:39:17.110940Z","shell.execute_reply.started":"2021-06-10T13:38:06.724768Z","shell.execute_reply":"2021-06-10T13:39:17.109982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Top cụm 3 từ xuất hiện nhiều trong tập dữ liệu đã Tiền xử lý\ntop_tri_grams = get_top_ngram(preprocessed_train['preprocessed_train_text'],n=3)\nx,y = map(list,zip(*top_tri_grams)) \nsns.barplot(x=y,y=x).set(title='Các cụm 3 từ xuất hiện nhiều nhất trong dữ liệu đã tiền xử lý')","metadata":{"execution":{"iopub.status.busy":"2021-06-10T13:39:17.112402Z","iopub.execute_input":"2021-06-10T13:39:17.112685Z","iopub.status.idle":"2021-06-10T13:40:39.923261Z","shell.execute_reply.started":"2021-06-10T13:39:17.112658Z","shell.execute_reply":"2021-06-10T13:40:39.921982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy là sau khi tiền xử lý, các từ, cụm 2 từ xuất hiện nhiều không còn là các từ, cụm từ mang tính khởi đầu của một câu hỏi như 'What', 'I', 'How', 'Why', 'What is', 'is the','What is the', 'What are the' ... nữa.  \n \nĐiều này là rất quan trọng trong bước feature extraction.  ","metadata":{}},{"cell_type":"markdown","source":"------------------------------------------------------------------------------------------------------------------------------------------------------------------","metadata":{}},{"cell_type":"markdown","source":"Tiếp theo sẽ là phần Feature Extraction  \n...","metadata":{}}]}