{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Quora Insincere Questions Classification","metadata":{"id":"P6TzjzwJ5YuM"}},{"cell_type":"markdown","source":"## 1. Loading Libraries","metadata":{"id":"IibKbRza5SeP"}},{"cell_type":"code","source":"!pip install regex eli5 emojic","metadata":{"id":"OICt4M8u5SeQ","execution":{"iopub.status.busy":"2021-06-09T14:45:30.734756Z","iopub.execute_input":"2021-06-09T14:45:30.736233Z","iopub.status.idle":"2021-06-09T14:48:00.477108Z","shell.execute_reply.started":"2021-06-09T14:45:30.736121Z","shell.execute_reply":"2021-06-09T14:48:00.476149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport os\nimport re\nimport csv\nimport string\nimport emoji\nimport regex\nimport eli5\nimport pickle\nimport gensim\nimport spacy\nimport gc\nfrom tqdm import tqdm\nimport random\nimport sklearn\n\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom collections import Counter\nimport matplotlib.pyplot as plt\nfrom wordcloud import WordCloud, STOPWORDS\nfrom scipy.sparse import hstack\nfrom IPython.display import Image\nfrom prettytable import PrettyTable\n\nfrom tqdm import tqdm_notebook\ntqdm_notebook().pandas()\n\nfrom nltk.stem import PorterStemmer, SnowballStemmer, WordNetLemmatizer\nfrom nltk.stem.lancaster import LancasterStemmer\nfrom nltk.util import ngrams\n\nfrom sklearn.metrics import confusion_matrix, log_loss\nfrom sklearn.model_selection import train_test_split, KFold, GridSearchCV, RandomizedSearchCV\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics import f1_score, classification_report\nfrom sklearn.calibration import CalibratedClassifierCV","metadata":{"id":"isqWlj_p5SeR","outputId":"a73b94d8-fa41-4144-8ee3-9fcdab9bcc43","execution":{"iopub.status.busy":"2021-06-09T14:48:00.478948Z","iopub.execute_input":"2021-06-09T14:48:00.479298Z","iopub.status.idle":"2021-06-09T14:48:09.750740Z","shell.execute_reply.started":"2021-06-09T14:48:00.479255Z","shell.execute_reply":"2021-06-09T14:48:09.749762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nnltk.download('wordnet')\nnltk.download('punkt')","metadata":{"id":"SfAGU_nb5SeU","execution":{"iopub.status.busy":"2021-06-09T14:48:09.753639Z","iopub.execute_input":"2021-06-09T14:48:09.754009Z","iopub.status.idle":"2021-06-09T14:48:49.825055Z","shell.execute_reply.started":"2021-06-09T14:48:09.753970Z","shell.execute_reply":"2021-06-09T14:48:49.824274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Loading data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ndf_test = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')\n\nprint(\"Number of data points in training data:\", df_train.shape[0])\nprint(\"Number of data points in test data:\", df_test.shape[0])","metadata":{"id":"z-wv-iqL4-2z","outputId":"0f4cbe4c-0570-4aae-da5f-f755f43797c9","execution":{"iopub.status.busy":"2021-06-09T14:48:49.826709Z","iopub.execute_input":"2021-06-09T14:48:49.827073Z","iopub.status.idle":"2021-06-09T14:48:53.978884Z","shell.execute_reply.started":"2021-06-09T14:48:49.827034Z","shell.execute_reply":"2021-06-09T14:48:53.977489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"id":"LI_tKBRK4-24","outputId":"d9689e4f-fe97-42e9-b1a5-6e189a50e4d6","execution":{"iopub.status.busy":"2021-06-09T14:48:53.980136Z","iopub.execute_input":"2021-06-09T14:48:53.980470Z","iopub.status.idle":"2021-06-09T14:48:53.998634Z","shell.execute_reply.started":"2021-06-09T14:48:53.980435Z","shell.execute_reply":"2021-06-09T14:48:53.997479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['question_text'].isnull().sum(), df_test['question_text'].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:48:53.999947Z","iopub.execute_input":"2021-06-09T14:48:54.000365Z","iopub.status.idle":"2021-06-09T14:48:54.143572Z","shell.execute_reply.started":"2021-06-09T14:48:54.000330Z","shell.execute_reply":"2021-06-09T14:48:54.142650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Nhận xét: \nTập dữ liệu huấn luyện gồm các cột: qid, question_text và target (giá trị nhị phân). Tất cả các quan sát là duy nhất với các giá trị khác null.","metadata":{}},{"cell_type":"markdown","source":"## 3. Data Analysis","metadata":{}},{"cell_type":"markdown","source":"### 3.1 Distribution of data points among output class","metadata":{}},{"cell_type":"code","source":"# Bar chart\nplt.subplot(1, 2, 1)\ndf_train.groupby('target')['qid'].count().plot.bar()\nplt.grid(True)\nplt.title('Target Count')\nplt.subplots_adjust(right=1.9)\n\n# Pie Chart\nplt.subplot(1, 2, 2)\nvalues = [df_train[df_train['target']==0].shape[0], df_train[df_train['target']==1].shape[0]]\nlabels = ['Sincere questions', 'Insincere questions']\n\nplt.pie(values, labels=labels, autopct='%1.1f%%', shadow=True)\nplt.title('Target Distribution')\nplt.tight_layout()\nplt.subplots_adjust(right=1.9)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:48:54.144795Z","iopub.execute_input":"2021-06-09T14:48:54.145148Z","iopub.status.idle":"2021-06-09T14:48:54.699012Z","shell.execute_reply.started":"2021-06-09T14:48:54.145114Z","shell.execute_reply":"2021-06-09T14:48:54.698255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Nhận xét: \n- Tập dữ liệu rất mất cân bằng với chỉ 6,2% insincere questions.\n- Nên sử dụng F1-score vì sự mất cân bằng trong dữ liệu.","metadata":{}},{"cell_type":"markdown","source":"### 3.2 Word cloud for both sincere and insincere questions","metadata":{}},{"cell_type":"code","source":"def display_wordcloud(data, title):\n    words_list = data.unique().tolist()\n    words = ' '.join(words_list)\n    \n    wordcloud = WordCloud(width = 800, height = 400,\n                      stopwords = set(STOPWORDS)).generate(words)\n\n    plt.figure(figsize=(20, 12), facecolor=None)\n    plt.imshow(wordcloud)\n    plt.title(f'Words in {title}')\n    plt.axis(\"off\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:48:54.701689Z","iopub.execute_input":"2021-06-09T14:48:54.702013Z","iopub.status.idle":"2021-06-09T14:48:54.707436Z","shell.execute_reply.started":"2021-06-09T14:48:54.701981Z","shell.execute_reply":"2021-06-09T14:48:54.706452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Wordcloud for Sincere Questions\ndisplay_wordcloud(df_train[df_train['target']==0]['question_text'], 'Sincere Questions')","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:48:54.709498Z","iopub.execute_input":"2021-06-09T14:48:54.710043Z","iopub.status.idle":"2021-06-09T14:49:45.636533Z","shell.execute_reply.started":"2021-06-09T14:48:54.710006Z","shell.execute_reply":"2021-06-09T14:49:45.635590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Wordcloud for Insincere Questions\ndisplay_wordcloud(df_train[df_train['target']==1]['question_text'], 'Insincere Questions')","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:49:45.637593Z","iopub.execute_input":"2021-06-09T14:49:45.637968Z","iopub.status.idle":"2021-06-09T14:49:51.537715Z","shell.execute_reply.started":"2021-06-09T14:49:45.637920Z","shell.execute_reply":"2021-06-09T14:49:51.536795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Nhận xét:\n - Những câu hỏi thiếu chân thành chứa nhiều từ ngữ xúc phạm.\n - Hầu hết các câu hỏi liên quan đến Con người, Hồi giáo, Phụ nữ, Trump, v.v.","metadata":{}},{"cell_type":"markdown","source":"### 3.4 Basic Feature Extraction","metadata":{"id":"-o2KqO-cG2YD"}},{"cell_type":"markdown","source":"#### Vấn đề:  Làm thế nào để có thể phân tích dữ liệu câu hỏi hiệu quả hơn theo từng yếu tố của câu như số từ, số ký tự, số ký tự viết hoa,... để từ đó có thể tìm ra đặc trưng của các câu hỏi sincere và insincere?","metadata":{}},{"cell_type":"markdown","source":"#### Giải pháp: \n- Chia các câu thành các từ hay tokenization.","metadata":{}},{"cell_type":"code","source":"# https://www.kaggle.com/sudalairajkumar/simple-exploration-notebook-qiqc\n\n# Number of words\ndf_train['num_words'] = df_train['question_text'].apply(lambda x: len(str(x).split()))\ndf_test['num_words'] = df_test['question_text'].apply(lambda x: len(str(x).split()))\n\n# Number of capital_letters\ndf_train['num_capital_let'] = df_train['question_text'].apply(lambda x: len([c for c in str(x) if c.isupper()]))\ndf_test['num_capital_let'] = df_test['question_text'].apply(lambda x: len([c for c in str(x) if c.isupper()]))\n\n# Number of special characters\ndf_train['num_special_char'] = df_train['question_text'].str.findall(r'[^a-zA-Z0-9 ]').str.len()\ndf_test['num_special_char'] = df_test['question_text'].str.findall(r'[^a-zA-Z0-9 ]').str.len()\n\n# Number of unique words\ndf_train['num_unique_words'] = df_train['question_text'].apply(lambda x: len(set(str(x).split())))\ndf_test['num_unique_words'] = df_test['question_text'].apply(lambda x: len(set(str(x).split())))\n\n# Number of numerics\ndf_train['num_numerics'] = df_train['question_text'].apply(lambda x: sum(c.isdigit() for c in x))\ndf_test['num_numerics'] = df_test['question_text'].apply(lambda x: sum(c.isdigit() for c in x))\n\n# Number of characters\ndf_train['num_char'] = df_train['question_text'].apply(lambda x: len(str(x)))\ndf_test['num_char'] = df_test['question_text'].apply(lambda x: len(str(x)))\n\n# Number of stopwords\ndf_train['num_stopwords'] = df_train['question_text'].apply(lambda x: len([c for c in str(x).lower().split() if c in STOPWORDS]))\ndf_test['num_stopwords'] = df_test['question_text'].apply(lambda x: len([c for c in str(x).lower().split() if c in STOPWORDS]))\n\ndf_train.head()","metadata":{"id":"-NF51YEpHI41","outputId":"c4e6c4fc-609c-44a9-cf39-d787a40d04c9","execution":{"iopub.status.busy":"2021-06-09T14:49:51.539098Z","iopub.execute_input":"2021-06-09T14:49:51.539407Z","iopub.status.idle":"2021-06-09T14:50:28.121479Z","shell.execute_reply.started":"2021-06-09T14:49:51.539376Z","shell.execute_reply":"2021-06-09T14:50:28.120677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Nhận xét:\n- Từ đó ta có thể phân tích các câu hỏi theo các đặc trưng mã được tokenization","metadata":{}},{"cell_type":"markdown","source":"### 3.5 Analysis on extracted features.","metadata":{}},{"cell_type":"code","source":"print(\"Minimum length of a question:\", min(df_train['num_words']))\nprint(\"Maximum length of a question:\", max(df_train['num_words']))","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:50:28.122705Z","iopub.execute_input":"2021-06-09T14:50:28.123046Z","iopub.status.idle":"2021-06-09T14:50:28.403538Z","shell.execute_reply.started":"2021-06-09T14:50:28.123010Z","shell.execute_reply":"2021-06-09T14:50:28.402716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_boxplot(_x, _y, _data, _title):\n    sns.boxplot(x=_x, y=_y, data=_data)\n    plt.grid(True)\n    plt.title(_title)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:50:28.404748Z","iopub.execute_input":"2021-06-09T14:50:28.405241Z","iopub.status.idle":"2021-06-09T14:50:28.420122Z","shell.execute_reply.started":"2021-06-09T14:50:28.405204Z","shell.execute_reply":"2021-06-09T14:50:28.419278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Boxplot: Number of words\nplt.subplot(2, 3, 1)\ndisplay_boxplot('target', 'num_words', df_train, 'words')\n\n# Boxplot: Number of chars\nplt.subplot(2, 3, 2)\ndisplay_boxplot('target', 'num_char', df_train, 'characters')\n\n# Boxplot: Number of unique words\nplt.subplot(2, 3, 3)\ndisplay_boxplot('target', 'num_unique_words', df_train, 'unique words')\n\n# Boxplot: Number of special characters\nplt.subplot(2, 3, 4)\ndisplay_boxplot('target', 'num_special_char', df_train, 'special characters')\n\n# Boxplot: Number of stopwords\nplt.subplot(2, 3, 5)\ndisplay_boxplot('target', 'num_stopwords', df_train, 'stopwords')\n\n# Boxplot: Number of capital letters\nplt.subplot(2, 3, 6)\ndisplay_boxplot('target', 'num_capital_let', df_train, 'capital letters')\n\n\nplt.subplots_adjust(right=3.0)\nplt.subplots_adjust(top=2.0)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:50:28.421842Z","iopub.execute_input":"2021-06-09T14:50:28.422267Z","iopub.status.idle":"2021-06-09T14:50:30.209310Z","shell.execute_reply.started":"2021-06-09T14:50:28.422222Z","shell.execute_reply":"2021-06-09T14:50:30.208523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation matrix\nf, ax = plt.subplots(figsize=(10, 8))\ncorr = df_train.corr()\nsns.heatmap(corr, ax=ax)\nplt.title(\"Correlation matrix\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:50:30.210533Z","iopub.execute_input":"2021-06-09T14:50:30.210853Z","iopub.status.idle":"2021-06-09T14:50:30.804112Z","shell.execute_reply.started":"2021-06-09T14:50:30.210817Z","shell.execute_reply":"2021-06-09T14:50:30.803330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Nhận xét:\n- Insincere questions có vẻ gồm nhiều từ và ký tự hơn.\n- Insincere questions có nhiều unique_words hơn sincere questions.","metadata":{}},{"cell_type":"code","source":"# Questions with most number of non-alphanumeric characters.\n  # Insincere questions are printed in red color.\n\nqids = df_train.sort_values('num_special_char', ascending=False)['qid'].head(20).values\nfor id in qids:\n  row = df_train[df_train['qid'].values == id]\n  if row['target'].values[0] == 1: \n    color = '\\033[31m'\n  else:\n    color = '\\033[0m'\n  print(color, row['question_text'].values[0], '\\n')","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:50:30.805364Z","iopub.execute_input":"2021-06-09T14:50:30.805711Z","iopub.status.idle":"2021-06-09T14:50:32.136787Z","shell.execute_reply.started":"2021-06-09T14:50:30.805675Z","shell.execute_reply":"2021-06-09T14:50:32.136025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Nhận xét:\n- Các câu hỏi về toán học hầu hết đều được phân loại là insincere questions, trong các câu hỏi này thường chứa các ký tự đặc biệt và các chữ số.\n- Một số câu hỏi cũng chứa biểu tượng cảm xúc và các ký tự không phải tiếng Anh.\n- Sự hiện diện của các dấu chấm câu có thể tăng thêm giá trị cho các mô hình ML.","metadata":{}},{"cell_type":"markdown","source":"## 4. Data Preprocessing and cleaning","metadata":{}},{"cell_type":"markdown","source":"#### Vấn đề: \n    - Văn bản khi chưa xử lý chứa các ký tự, từ ngữ không có ích cho việc đánh giá đó là câu hỏi sincere or insincere\n    - Xử lý văn bản còn giúp cải thiện F1-score","metadata":{}},{"cell_type":"markdown","source":"#### Giải pháp:\n- Thay thế các phương trình toán học và url bằng cách viết tắt phổ biến.\n- Sửa các từ viết tắt.\n- Sửa lỗi chính tả.\n- Bỏ dấu chấm câu, các kí tự đặc biệt.\n- Loại bỏ các stopword.\n- Sử dụng WordNet Lemmatizer để chuyển từ về dạng gốc của nó (kiểu teacher, teaches thành teach; best, better thành good,..)","metadata":{"id":"S-HsERsXQ1mU"}},{"cell_type":"code","source":"# Replacing math equations and url addresses with tags.\n# https://www.kaggle.com/canming/ensemble-mean-iii-64-36\ndef clean_tag(x):\n  if '[math]' in x:\n    x = re.sub('\\[math\\].*?math\\]', 'MATH EQUATION', x) #replacing with [MATH EQUATION]\n    \n  if 'http' in x or 'www' in x:\n    x = re.sub('(?:(?:https?|ftp):\\/\\/)?[\\w/\\-?=%.]+\\.[\\w/\\-?=%.]+', 'URL', x) #replacing with [url]\n  return x","metadata":{"id":"GH6BniD7ABhN","execution":{"iopub.status.busy":"2021-06-09T14:50:32.139478Z","iopub.execute_input":"2021-06-09T14:50:32.139730Z","iopub.status.idle":"2021-06-09T14:50:32.144042Z","shell.execute_reply.started":"2021-06-09T14:50:32.139703Z","shell.execute_reply":"2021-06-09T14:50:32.143047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clean_punct\n# https://www.kaggle.com/canming/ensemble-mean-iii-64-36\n\npuncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', \n        '•', '~', '@', '£', '·', '_', '{', '}', '©', '^', '®', '`', '<', '→', '°', '€', '™', '›', '♥', '←', '×', '§', '″', '′', \n        '█', '…', '“', '★', '”', '–', '●', '►', '−', '¢', '¬', '░', '¡', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', \n        '—', '‹', '─', '▒', '：', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', '¯', '♦', '¤', '▲', '¸', '⋅', '‘', '∞', \n        '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '・', '╦', '╣', '╔', '╗', '▬', '❤', '≤', '‡', '√', '◄', '━', \n        '⇒', '▶', '≥', '╝', '♡', '◊', '。', '✈', '≡', '☺', '✔', '↵', '≈', '✓', '♣', '☎', '℃', '◦', '└', '‟', '～', '！', '○', \n        '◆', '№', '♠', '▌', '✿', '▸', '⁄', '□', '❖', '✦', '．', '÷', '｜', '┃', '／', '￥', '╠', '↩', '✭', '▐', '☼', '☻', '┐', \n        '├', '«', '∼', '┌', '℉', '☮', '฿', '≦', '♬', '✧', '〉', '－', '⌂', '✖', '･', '◕', '※', '‖', '◀', '‰', '\\x97', '↺', \n        '∆', '┘', '┬', '╬', '،', '⌘', '⊂', '＞', '〈', '⎙', '？', '☠', '⇐', '▫', '∗', '∈', '≠', '♀', '♔', '˚', '℗', '┗', '＊', \n        '┼', '❀', '＆', '∩', '♂', '‿', '∑', '‣', '➜', '┛', '⇓', '☯', '⊖', '☀', '┳', '；', '∇', '⇑', '✰', '◇', '♯', '☞', '´', \n        '↔', '┏', '｡', '◘', '∂', '✌', '♭', '┣', '┴', '┓', '✨', '\\xa0', '˜', '❥', '┫', '℠', '✒', '［', '∫', '\\x93', '≧', '］', \n        '\\x94', '∀', '♛', '\\x96', '∨', '◎', '↻', '⇩', '＜', '≫', '✩', '✪', '♕', '؟', '₤', '☛', '╮', '␊', '＋', '┈', '％', \n        '╋', '▽', '⇨', '┻', '⊗', '￡', '।', '▂', '✯', '▇', '＿', '➤', '✞', '＝', '▷', '△', '◙', '▅', '✝', '∧', '␉', '☭', \n        '┊', '╯', '☾', '➔', '∴', '\\x92', '▃', '↳', '＾', '׳', '➢', '╭', '➡', '＠', '⊙', '☢', '˝', '∏', '„', '∥', '❝', '☐', \n        '▆', '╱', '⋙', '๏', '☁', '⇔', '▔', '\\x91', '➚', '◡', '╰', '\\x85', '♢', '˙', '۞', '✘', '✮', '☑', '⋆', 'ⓘ', '❒', \n        '☣', '✉', '⌊', '➠', '∣', '❑', '◢', 'ⓒ', '\\x80', '〒', '∕', '▮', '⦿', '✫', '✚', '⋯', '♩', '☂', '❞', '‗', '܂', '☜', \n        '‾', '✜', '╲', '∘', '⟩', '＼', '⟨', '·', '✗', '♚', '∅', 'ⓔ', '◣', '͡', '‛', '❦', '◠', '✄', '❄', '∃', '␣', '≪', '｢', \n        '≅', '◯', '☽', '∎', '｣', '❧', '̅', 'ⓐ', '↘', '⚓', '▣', '˘', '∪', '⇢', '✍', '⊥', '＃', '⎯', '↠', '۩', '☰', '◥', \n        '⊆', '✽', '⚡', '↪', '❁', '☹', '◼', '☃', '◤', '❏', 'ⓢ', '⊱', '➝', '̣', '✡', '∠', '｀', '▴', '┤', '∝', '♏', 'ⓐ', \n        '✎', ';', '␤', '＇', '❣', '✂', '✤', 'ⓞ', '☪', '✴', '⌒', '˛', '♒', '＄', '✶', '▻', 'ⓔ', '◌', '◈', '❚', '❂', '￦', \n        '◉', '╜', '̃', '✱', '╖', '❉', 'ⓡ', '↗', 'ⓣ', '♻', '➽', '׀', '✲', '✬', '☉', '▉', '≒', '☥', '⌐', '♨', '✕', 'ⓝ', \n        '⊰', '❘', '＂', '⇧', '̵', '➪', '▁', '▏', '⊃', 'ⓛ', '‚', '♰', '́', '✏', '⏑', '̶', 'ⓢ', '⩾', '￠', '❍', '≃', '⋰', '♋', \n        '､', '̂', '❋', '✳', 'ⓤ', '╤', '▕', '⌣', '✸', '℮', '⁺', '▨', '╨', 'ⓥ', '♈', '❃', '☝', '✻', '⊇', '≻', '♘', '♞', \n        '◂', '✟', '⌠', '✠', '☚', '✥', '❊', 'ⓒ', '⌈', '❅', 'ⓡ', '♧', 'ⓞ', '▭', '❱', 'ⓣ', '∟', '☕', '♺', '∵', '⍝', 'ⓑ', \n        '✵', '✣', '٭', '♆', 'ⓘ', '∶', '⚜', '◞', '்', '✹', '➥', '↕', '̳', '∷', '✋', '➧', '∋', '̿', 'ͧ', '┅', '⥤', '⬆', '⋱', \n        '☄', '↖', '⋮', '۔', '♌', 'ⓛ', '╕', '♓', '❯', '♍', '▋', '✺', '⭐', '✾', '♊', '➣', '▿', 'ⓑ', '♉', '⏠', '◾', '▹', \n        '⩽', '↦', '╥', '⍵', '⌋', '։', '➨', '∮', '⇥', 'ⓗ', 'ⓓ', '⁻', '⎝', '⌥', '⌉', '◔', '◑', '✼', '♎', '♐', '╪', '⊚', \n        '☒', '⇤', 'ⓜ', '⎠', '◐', '⚠', '╞', '◗', '⎕', 'ⓨ', '☟', 'ⓟ', '♟', '❈', '↬', 'ⓓ', '◻', '♮', '❙', '♤', '∉', '؛', \n        '⁂', 'ⓝ', '־', '♑', '╫', '╓', '╳', '⬅', '☔', '☸', '┄', '╧', '׃', '⎢', '❆', '⋄', '⚫', '̏', '☏', '➞', '͂', '␙', \n        'ⓤ', '◟', '̊', '⚐', '✙', '↙', '̾', '℘', '✷', '⍺', '❌', '⊢', '▵', '✅', 'ⓖ', '☨', '▰', '╡', 'ⓜ', '☤', '∽', '╘', \n        '˹', '↨', '♙', '⬇', '♱', '⌡', '⠀', '╛', '❕', '┉', 'ⓟ', '̀', '♖', 'ⓚ', '┆', '⎜', '◜', '⚾', '⤴', '✇', '╟', '⎛', \n        '☩', '➲', '➟', 'ⓥ', 'ⓗ', '⏝', '◃', '╢', '↯', '✆', '˃', '⍴', '❇', '⚽', '╒', '̸', '♜', '☓', '➳', '⇄', '☬', '⚑', \n        '✐', '⌃', '◅', '▢', '❐', '∊', '☈', '॥', '⎮', '▩', 'ு', '⊹', '‵', '␔', '☊', '➸', '̌', '☿', '⇉', '⊳', '╙', 'ⓦ', \n        '⇣', '｛', '̄', '↝', '⎟', '▍', '❗', '״', '΄', '▞', '◁', '⛄', '⇝', '⎪', '♁', '⇠', '☇', '✊', 'ி', '｝', '⭕', '➘', \n        '⁀', '☙', '❛', '❓', '⟲', '⇀', '≲', 'ⓕ', '⎥', '\\u06dd', 'ͤ', '₋', '̱', '̎', '♝', '≳', '▙', '➭', '܀', 'ⓖ', '⇛', '▊', \n        '⇗', '̷', '⇱', '℅', 'ⓧ', '⚛', '̐', '̕', '⇌', '␀', '≌', 'ⓦ', '⊤', '̓', '☦', 'ⓕ', '▜', '➙', 'ⓨ', '⌨', '◮', '☷', \n        '◍', 'ⓚ', '≔', '⏩', '⍳', '℞', '┋', '˻', '▚', '≺', 'ْ', '▟', '➻', '̪', '⏪', '̉', '⎞', '┇', '⍟', '⇪', '▎', '⇦', '␝', \n        '⤷', '≖', '⟶', '♗', '̴', '♄', 'ͨ', '̈', '❜', '̡', '▛', '✁', '➩', 'ா', '˂', '↥', '⏎', '⎷', '̲', '➖', '↲', '⩵', '̗', '❢', \n        '≎', '⚔', '⇇', '̑', '⊿', '̖', '☍', '➹', '⥊', '⁁', '✢']\n\ndef clean_punct(x):\n  x = str(x)\n  for punct in puncts:\n    if punct in x:\n      x = x.replace(punct, ' ')\n    return x","metadata":{"id":"yGinDy5763K-","execution":{"iopub.status.busy":"2021-06-09T14:50:32.145162Z","iopub.execute_input":"2021-06-09T14:50:32.145487Z","iopub.status.idle":"2021-06-09T14:50:32.179818Z","shell.execute_reply.started":"2021-06-09T14:50:32.145454Z","shell.execute_reply":"2021-06-09T14:50:32.178683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# correct_mispell\n# https://www.kaggle.com/oysiyl/107-place-solution-using-public-kernel\n\nmispell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling', 'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2', 'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist', 'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', 'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'bitcoin', 'narcissit': 'narcissist', 'bigdata': 'big data', '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', 'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization', \n                'electroneum':'bitcoin','nanodegree':'degree','hotstar':'star','dream11':'dream','ftre':'fire','tensorflow':'framework','unocoin':'bitcoin',\n                'lnmiit':'limit','unacademy':'academy','altcoin':'bitcoin','altcoins':'bitcoin','litecoin':'bitcoin','coinbase':'bitcoin','cryptocurency':'cryptocurrency',\n                'simpliv':'simple','quoras':'quora','schizoids':'psychopath','remainers':'remainder','twinflame':'soulmate','quorans':'quora','brexit':'demonetized',\n                'iiest':'institute','dceu':'comics','pessat':'exam','uceed':'college','bhakts':'devotee','boruto':'anime',\n                'cryptocoin':'bitcoin','blockchains':'blockchain','fiancee':'fiance','redmi':'smartphone','oneplus':'smartphone','qoura':'quora','deepmind':'framework','ryzen':'cpu','whattsapp':'whatsapp',\n                'undertale':'adventure','zenfone':'smartphone','cryptocurencies':'cryptocurrencies','koinex':'bitcoin','zebpay':'bitcoin','binance':'bitcoin','whtsapp':'whatsapp',\n                'reactjs':'framework','bittrex':'bitcoin','bitconnect':'bitcoin','bitfinex':'bitcoin','yourquote':'your quote','whyis':'why is','jiophone':'smartphone',\n                'dogecoin':'bitcoin','onecoin':'bitcoin','poloniex':'bitcoin','7700k':'cpu','angular2':'framework','segwit2x':'bitcoin','hashflare':'bitcoin','940mx':'gpu',\n                'openai':'framework','hashflare':'bitcoin','1050ti':'gpu','nearbuy':'near buy','freebitco':'bitcoin','antminer':'bitcoin','filecoin':'bitcoin','whatapp':'whatsapp',\n                'empowr':'empower','1080ti':'gpu','crytocurrency':'cryptocurrency','8700k':'cpu','whatsaap':'whatsapp','g4560':'cpu','payymoney':'pay money',\n                'fuckboys':'fuck boys','intenship':'internship','zcash':'bitcoin','demonatisation':'demonetization','narcicist':'narcissist','mastuburation':'masturbation',\n                'trignometric':'trigonometric','cryptocurreny':'cryptocurrency','howdid':'how did','crytocurrencies':'cryptocurrencies','phycopath':'psychopath',\n                'bytecoin':'bitcoin','possesiveness':'possessiveness','scollege':'college','humanties':'humanities','altacoin':'bitcoin','demonitised':'demonetized',\n                'brasília':'brazilia','accolite':'accolyte','econimics':'economics','varrier':'warrier','quroa':'quora','statergy':'strategy','langague':'language',\n                'splatoon':'game','7600k':'cpu','gate2018':'gate 2018','in2018':'in 2018','narcassist':'narcissist','jiocoin':'bitcoin','hnlu':'hulu','7300hq':'cpu',\n                'weatern':'western','interledger':'blockchain','deplation':'deflation', 'cryptocurrencies':'cryptocurrency', 'bitcoin':'blockchain cryptocurrency',}\n\ndef correct_mispell(x):\n  words = x.split()\n  for i in range(0, len(words)):\n    if mispell_dict.get(words[i]) is not None:\n      words[i] = mispell_dict.get(words[i])\n    elif mispell_dict.get(words[i].lower()) is not None:\n      words[i] = mispell_dict.get(words[i].lower())\n        \n  words = \" \".join(words)\n  return words","metadata":{"id":"aiEnsv5363Hn","execution":{"iopub.status.busy":"2021-06-09T14:50:32.181240Z","iopub.execute_input":"2021-06-09T14:50:32.181574Z","iopub.status.idle":"2021-06-09T14:50:32.199322Z","shell.execute_reply.started":"2021-06-09T14:50:32.181542Z","shell.execute_reply":"2021-06-09T14:50:32.198490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove stopwords\ndef remove_stopwords(x):\n  x = [word for word in x.split() if word not in STOPWORDS]\n  x = ' '.join(x)\n\n  return x","metadata":{"id":"gG0IuCDL8CXZ","execution":{"iopub.status.busy":"2021-06-09T14:50:32.200325Z","iopub.execute_input":"2021-06-09T14:50:32.200780Z","iopub.status.idle":"2021-06-09T14:50:32.211662Z","shell.execute_reply.started":"2021-06-09T14:50:32.200742Z","shell.execute_reply":"2021-06-09T14:50:32.210892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clean word contractions\n## https://www.kaggle.com/theoviel/improve-your-score-with-text-preprocessing-v2 \n\ncontraction_mapping = {\"We'd\": \"We had\", \"That'd\": \"That had\", \"AREN'T\": \"Are not\", \"HADN'T\": \"Had not\", \"Could've\": \"Could have\", \"LeT's\": \"Let us\", \"How'll\": \"How will\", \"They'll\": \"They will\", \"DOESN'T\": \"Does not\", \"HE'S\": \"He has\", \"O'Clock\": \"Of the clock\", \"Who'll\": \"Who will\", \"What'S\": \"What is\", \"Ain't\": \"Am not\", \"WEREN'T\": \"Were not\", \"Y'all\": \"You all\", \"Y'ALL\": \"You all\", \"Here's\": \"Here is\", \"It'd\": \"It had\", \"Should've\": \"Should have\", \"I'M\": \"I am\", \"ISN'T\": \"Is not\", \"Would've\": \"Would have\", \"He'll\": \"He will\", \"DON'T\": \"Do not\", \"She'd\": \"She had\", \"WOULDN'T\": \"Would not\", \"She'll\": \"She will\", \"IT's\": \"It is\", \"There'd\": \"There had\", \"It'll\": \"It will\", \"You'll\": \"You will\", \"He'd\": \"He had\", \"What'll\": \"What will\", \"Ma'am\": \"Madam\", \"CAN'T\": \"Can not\", \"THAT'S\": \"That is\", \"You've\": \"You have\", \"She's\": \"She is\", \"Weren't\": \"Were not\", \"They've\": \"They have\", \"Couldn't\": \"Could not\", \"When's\": \"When is\", \"Haven't\": \"Have not\", \"We'll\": \"We will\", \"That's\": \"That is\", \"We're\": \"We are\", \"They're\": \"They' are\", \"You'd\": \"You would\", \"How'd\": \"How did\", \"What're\": \"What are\", \"Hasn't\": \"Has not\", \"Wasn't\": \"Was not\", \"Won't\": \"Will not\", \"There's\": \"There is\", \"Didn't\": \"Did not\", \"Doesn't\": \"Does not\", \"You're\": \"You are\", \"He's\": \"He is\", \"SO's\": \"So is\", \"We've\": \"We have\", \"Who's\": \"Who is\", \"Wouldn't\": \"Would not\", \"Why's\": \"Why is\", \"WHO's\": \"Who is\", \"Let's\": \"Let us\", \"How's\": \"How is\", \"Can't\": \"Can not\", \"Where's\": \"Where is\", \"They'd\": \"They had\", \"Don't\": \"Do not\", \"Shouldn't\":\"Should not\", \"Aren't\":\"Are not\", \"ain't\": \"is not\", \"What's\": \"What is\", \"It's\": \"It is\", \"Isn't\":\"Is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"'cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\",  \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\",  \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\",  \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\",  \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\" }\n\ndef clean_contractions(text):\n    specials = [\"’\", \"‘\", \"´\", \"`\"]\n    for s in specials:\n        text = text.replace(s, \"'\")\n    \n    text = ' '.join([contraction_mapping[t] if t in contraction_mapping else t for t in text.split(\" \")])\n    return text","metadata":{"id":"yUvNn9jW8sMX","execution":{"iopub.status.busy":"2021-06-09T14:50:32.214544Z","iopub.execute_input":"2021-06-09T14:50:32.214906Z","iopub.status.idle":"2021-06-09T14:50:32.233809Z","shell.execute_reply.started":"2021-06-09T14:50:32.214881Z","shell.execute_reply":"2021-06-09T14:50:32.232827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# word lemmatizing\n\nlemmatizer = WordNetLemmatizer()\ndef lemma_text(x):\n  x = x.split()\n  x = [lemmatizer.lemmatize(word) for word in x]\n  x = ' '.join(x)\n\n  return x","metadata":{"id":"NtFry3bK88Z_","execution":{"iopub.status.busy":"2021-06-09T14:50:32.235102Z","iopub.execute_input":"2021-06-09T14:50:32.235553Z","iopub.status.idle":"2021-06-09T14:50:32.244980Z","shell.execute_reply.started":"2021-06-09T14:50:32.235520Z","shell.execute_reply":"2021-06-09T14:50:32.244051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_cleaning(x):\n  x = clean_tag(x)\n  x = clean_punct(x)\n  x = correct_mispell(x)\n  x = remove_stopwords(x)\n  x = clean_contractions(x)\n  x = lemma_text(x)\n  return x","metadata":{"id":"gXBCN3Oo9dcq","execution":{"iopub.status.busy":"2021-06-09T14:50:32.249652Z","iopub.execute_input":"2021-06-09T14:50:32.250017Z","iopub.status.idle":"2021-06-09T14:50:32.254466Z","shell.execute_reply.started":"2021-06-09T14:50:32.249988Z","shell.execute_reply":"2021-06-09T14:50:32.253622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preprocessing given train and test data\ndf_train['preprocessed_question_text'] = df_train['question_text'].progress_map(lambda x: data_cleaning(x))\ndf_test['preprocessed_question_text'] = df_test['question_text'].progress_map(lambda x: data_cleaning(x))","metadata":{"id":"k8mloM359dSt","outputId":"32059bda-f63a-4ffc-838c-aa3fdca31695","execution":{"iopub.status.busy":"2021-06-09T14:50:32.256040Z","iopub.execute_input":"2021-06-09T14:50:32.256533Z","iopub.status.idle":"2021-06-09T14:52:05.049522Z","shell.execute_reply.started":"2021-06-09T14:50:32.256498Z","shell.execute_reply":"2021-06-09T14:52:05.048664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Kết quả: Dữ liệu sau khi được làm sạch \n#### Dữ liệu sau khi được làm sạch sẽ dử dụng TF-IDF để vector hóa rồi đưa vào model để train","metadata":{}},{"cell_type":"markdown","source":"## 5. Train, Test & Val split\n","metadata":{"id":"59xa5YkyP9JP"}},{"cell_type":"markdown","source":"Chia data Train ra thành 2 phần: Train và Val (Tập Test không có target nên phải cần 1 tập khác để kiểm tra máy đang học đúng hay sai)","metadata":{}},{"cell_type":"code","source":"split_ratio = 0.1\nX = df_train.drop('target', axis = 'columns')\ny = df_train['target']\n# y = df_train['target'].values\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=split_ratio, random_state=2019)\n\n# X_test = df_test.drop('target', axis = 'columns')\n# y_test = df_test['target']\nX_test = df_test\n\nprint('X_train: ', X_train.shape, y_train.shape)\n# print('X_test: ',X_test.shape, y_test.shape)\nprint('X_val: ',X_val.shape, y_val.shape)\nprint('X_test:', X_test.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T14:52:05.050819Z","iopub.execute_input":"2021-06-09T14:52:05.051214Z","iopub.status.idle":"2021-06-09T14:52:06.231248Z","shell.execute_reply.started":"2021-06-09T14:52:05.051177Z","shell.execute_reply":"2021-06-09T14:52:06.230436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check for percentage of class disb in train test split.\n\ndef plot_class_disb(class_disb, data, disb_name):\n  class_disb.plot(kind=\"bar\")\n  plt.xlabel('class')\n  plt.ylabel('Datapoints per class')\n  plt.title(f'Distribution of yi in {disb_name}')\n  plt.grid(True)\n  \n  sorted_yi = np.argsort(-class_disb.values)\n  print(disb_name,':')\n  for i in sorted_yi:\n    print('Number of data points in class', i, ':', class_disb.values[i], '(', np.round((class_disb.values[i]/data.shape[0]*100), 3), '%)')\n  \n  print('-'*50)","metadata":{"id":"WIkZQaViP9Jb","execution":{"iopub.status.busy":"2021-06-09T14:52:06.232514Z","iopub.execute_input":"2021-06-09T14:52:06.232841Z","iopub.status.idle":"2021-06-09T14:52:06.240525Z","shell.execute_reply.started":"2021-06-09T14:52:06.232804Z","shell.execute_reply":"2021-06-09T14:52:06.239240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Kiểm tra xem các class ở trong các file có cân bằng không. Tránh các trường hợp kiểu chia xong file train vs val mà train chỉ gồm mỗi 1 class hay val chỉ có mỗi 1 class","metadata":{}},{"cell_type":"markdown","source":"## 6. Baseline Model(LR) with basic extracted features","metadata":{"id":"RAO0b45ZQWZS"}},{"cell_type":"code","source":"tfidf = TfidfVectorizer(stop_words='english', ngram_range=(1, 3))\ntfidf.fit_transform(list(df_train['preprocessed_question_text'].values) + list(df_test['preprocessed_question_text'].values))\n\nX_train_ques = tfidf.transform(X_train['preprocessed_question_text'].values)\nX_test_ques = tfidf.transform(X_test['preprocessed_question_text'].values)\nX_val_ques = tfidf.transform(X_val['preprocessed_question_text'].values)\n\nprint(X_train_ques.shape)\nprint(X_test_ques.shape)\nprint(X_val_ques.shape)","metadata":{"id":"nNb8-zlPQWZd","outputId":"54cae894-0f46-44e8-a30f-72dc7e1dd8e8","execution":{"iopub.status.busy":"2021-06-09T14:52:06.241779Z","iopub.execute_input":"2021-06-09T14:52:06.242319Z","iopub.status.idle":"2021-06-09T14:54:36.780985Z","shell.execute_reply.started":"2021-06-09T14:52:06.242278Z","shell.execute_reply":"2021-06-09T14:54:36.780046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Kết quả: \n- TF-IDF đưa các text đã được preprocess về dạng vector . \n- Output của TF-IDF là ma trận trọng số của các word trong text. Chiều thứ nhất là số dòng (số data), chiều thứ 2 là độ dài vector sau khi chuyển từ text sang.","metadata":{}},{"cell_type":"code","source":"# Standardize stats features\nfrom sklearn.preprocessing import StandardScaler\n\n# number of words\nnum_words =  StandardScaler()\nX_train_num_words = num_words.fit_transform(X_train['num_words'].values.reshape(-1, 1))\nX_test_num_words = num_words.transform(X_test['num_words'].values.reshape(-1, 1))\nX_val_num_words = num_words.transform(X_val['num_words'].values.reshape(-1, 1))\n\n# number of unique words\nnum_unique_words =  StandardScaler()\nX_train_num_unique_words = num_unique_words.fit_transform(X_train['num_unique_words'].values.reshape(-1, 1))\nX_test_num_unique_words = num_unique_words.transform(X_test['num_unique_words'].values.reshape(-1, 1))\nX_val_num_unique_words = num_unique_words.transform(X_val['num_unique_words'].values.reshape(-1, 1))\n\n# number of char\nnum_char =  StandardScaler()\nX_train_num_char = num_char.fit_transform(X_train['num_char'].values.reshape(-1, 1))\nX_test_num_char = num_char.transform(X_test['num_char'].values.reshape(-1, 1))\nX_val_num_char = num_char.transform(X_val['num_char'].values.reshape(-1, 1))\n\n# number of stopwords\nnum_stopwords =  StandardScaler()\nX_train_num_stopwords = num_stopwords.fit_transform(X_train['num_stopwords'].values.reshape(-1, 1))\nX_test_num_stopwords = num_stopwords.transform(X_test['num_stopwords'].values.reshape(-1, 1))\nX_val_num_stopwords = num_stopwords.transform(X_val['num_stopwords'].values.reshape(-1, 1))","metadata":{"id":"jti8vx_fQWZo","execution":{"iopub.status.busy":"2021-06-09T14:54:36.782250Z","iopub.execute_input":"2021-06-09T14:54:36.782724Z","iopub.status.idle":"2021-06-09T14:54:36.863557Z","shell.execute_reply.started":"2021-06-09T14:54:36.782686Z","shell.execute_reply":"2021-06-09T14:54:36.862685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stacking features \n\nX_tr = hstack((\n    X_train_ques,\n    X_train_num_words,\n    X_train_num_unique_words,\n    X_train_num_char,\n    X_train_num_stopwords\n))\n\nX_te = hstack((\n    X_test_ques,\n    X_test_num_words,\n    X_test_num_unique_words,\n    X_test_num_char,\n    X_test_num_stopwords\n))\n\nX_cv = hstack((\n    X_val_ques,\n    X_val_num_words,\n    X_val_num_unique_words,\n    X_val_num_char,\n    X_val_num_stopwords\n))\n\nprint(X_tr.shape, y_train.shape)\n# print(X_te.shape, y_test.shape)\nprint(X_cv.shape, y_val.shape)","metadata":{"id":"ote82PP0QWZq","outputId":"956c0522-0bc3-4c74-fb6b-08d1e9edb213","execution":{"iopub.status.busy":"2021-06-09T14:54:36.864795Z","iopub.execute_input":"2021-06-09T14:54:36.865539Z","iopub.status.idle":"2021-06-09T14:54:37.677201Z","shell.execute_reply.started":"2021-06-09T14:54:36.865512Z","shell.execute_reply":"2021-06-09T14:54:37.676302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parmas = {'C': [0.001, 0.001, 0.1, 1, 10]}\n\ngridsearch = GridSearchCV(LogisticRegression(), parmas, scoring='f1', n_jobs=-1, verbose=1)\ngridsearch.fit(X_tr, y_train)","metadata":{"id":"KKKXGsGoQWZv","outputId":"c80b6ca3-d823-4145-e4e0-51b690981ebb","execution":{"iopub.status.busy":"2021-06-09T14:54:37.678413Z","iopub.execute_input":"2021-06-09T14:54:37.678991Z","iopub.status.idle":"2021-06-09T15:52:10.924609Z","shell.execute_reply.started":"2021-06-09T14:54:37.678950Z","shell.execute_reply":"2021-06-09T15:52:10.922690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gridsearch.best_params_","metadata":{"id":"AfoSIoeYY87g","outputId":"5b7c4462-b972-4f15-d0b1-7a8bffa6c274","execution":{"iopub.status.busy":"2021-06-09T15:52:10.927221Z","iopub.execute_input":"2021-06-09T15:52:10.927565Z","iopub.status.idle":"2021-06-09T15:52:10.939006Z","shell.execute_reply.started":"2021-06-09T15:52:10.927526Z","shell.execute_reply":"2021-06-09T15:52:10.937818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = LogisticRegression(C=1)\nclf.fit(X_tr, y_train)","metadata":{"id":"YdlZgQeiQWZ2","outputId":"31008baf-0b3d-4525-e849-d65e184ff3e7","execution":{"iopub.status.busy":"2021-06-09T15:52:10.940265Z","iopub.execute_input":"2021-06-09T15:52:10.940950Z","iopub.status.idle":"2021-06-09T15:56:07.670354Z","shell.execute_reply.started":"2021-06-09T15:52:10.940913Z","shell.execute_reply":"2021-06-09T15:56:07.669526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Prediction","metadata":{}},{"cell_type":"code","source":"y_pred_res = clf.predict(X_te)\nprint(len(y_pred_res))","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:56:07.671648Z","iopub.execute_input":"2021-06-09T15:56:07.672000Z","iopub.status.idle":"2021-06-09T15:56:07.884131Z","shell.execute_reply.started":"2021-06-09T15:56:07.671961Z","shell.execute_reply":"2021-06-09T15:56:07.883019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = X_test['qid'].to_dict()\nqid = []\nfor (key, val) in temp.items():\n    qid.append(val)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:56:07.885492Z","iopub.execute_input":"2021-06-09T15:56:07.886026Z","iopub.status.idle":"2021-06-09T15:56:08.148331Z","shell.execute_reply.started":"2021-06-09T15:56:07.885983Z","shell.execute_reply":"2021-06-09T15:56:08.147354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(qid[:4])","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:56:08.149685Z","iopub.execute_input":"2021-06-09T15:56:08.150040Z","iopub.status.idle":"2021-06-09T15:56:08.156066Z","shell.execute_reply.started":"2021-06-09T15:56:08.150002Z","shell.execute_reply":"2021-06-09T15:56:08.155155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\n\npb = tqdm(range(len(y_pred_res)))\nres = {}\n\nfor i in pb:\n    res[qid[i]] = y_pred_res[i]\n\n# print(res)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:56:08.157510Z","iopub.execute_input":"2021-06-09T15:56:08.158070Z","iopub.status.idle":"2021-06-09T15:56:08.535197Z","shell.execute_reply.started":"2021-06-09T15:56:08.158035Z","shell.execute_reply":"2021-06-09T15:56:08.534331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import csv\nwith open('submission.csv','w', newline='') as csv_file:\n    fieldnames = ['qid', 'prediction']\n    writer = csv.DictWriter(csv_file, fieldnames=fieldnames)\n\n    writer.writeheader()\n    for (key, val) in res.items():\n        writer.writerow({'qid': key, 'prediction': val})\n","metadata":{"execution":{"iopub.status.busy":"2021-06-09T15:56:08.536724Z","iopub.execute_input":"2021-06-09T15:56:08.537102Z","iopub.status.idle":"2021-06-09T15:56:10.039374Z","shell.execute_reply.started":"2021-06-09T15:56:08.537064Z","shell.execute_reply":"2021-06-09T15:56:10.038415Z"},"trusted":true},"execution_count":null,"outputs":[]}]}