{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-05-23T01:17:10.737184Z","iopub.execute_input":"2021-05-23T01:17:10.737617Z","iopub.status.idle":"2021-05-23T01:17:10.749437Z","shell.execute_reply.started":"2021-05-23T01:17:10.737534Z","shell.execute_reply":"2021-05-23T01:17:10.748154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-05-23T02:40:27.70539Z","iopub.execute_input":"2021-05-23T02:40:27.706035Z","iopub.status.idle":"2021-05-23T02:40:27.711627Z","shell.execute_reply.started":"2021-05-23T02:40:27.705989Z","shell.execute_reply":"2021-05-23T02:40:27.710353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2021-05-23T02:40:23.317471Z","iopub.execute_input":"2021-05-23T02:40:23.317871Z","iopub.status.idle":"2021-05-23T02:40:27.703651Z","shell.execute_reply.started":"2021-05-23T02:40:23.31784Z","shell.execute_reply":"2021-05-23T02:40:27.702778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T02:40:27.713009Z","iopub.execute_input":"2021-05-23T02:40:27.713656Z","iopub.status.idle":"2021-05-23T02:40:27.748293Z","shell.execute_reply.started":"2021-05-23T02:40:27.713611Z","shell.execute_reply":"2021-05-23T02:40:27.747104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:16.904654Z","iopub.execute_input":"2021-05-23T01:17:16.9051Z","iopub.status.idle":"2021-05-23T01:17:17.176484Z","shell.execute_reply.started":"2021-05-23T01:17:16.905054Z","shell.execute_reply":"2021-05-23T01:17:17.175395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:17.179305Z","iopub.execute_input":"2021-05-23T01:17:17.179603Z","iopub.status.idle":"2021-05-23T01:17:17.184825Z","shell.execute_reply.started":"2021-05-23T01:17:17.179572Z","shell.execute_reply":"2021-05-23T01:17:17.184104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1. EDA","metadata":{}},{"cell_type":"code","source":"train_df['question_text'].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:17.186123Z","iopub.execute_input":"2021-05-23T01:17:17.186647Z","iopub.status.idle":"2021-05-23T01:17:17.331439Z","shell.execute_reply.started":"2021-05-23T01:17:17.186604Z","shell.execute_reply":"2021-05-23T01:17:17.330374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['question_text'].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:17.332895Z","iopub.execute_input":"2021-05-23T01:17:17.333226Z","iopub.status.idle":"2021-05-23T01:17:17.473785Z","shell.execute_reply.started":"2021-05-23T01:17:17.333196Z","shell.execute_reply":"2021-05-23T01:17:17.472032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tập huấn luyện không có giá trị NULL.","metadata":{}},{"cell_type":"code","source":"# Đổi tên cột 'question_text' -> 'question' \ntrain_df = train_df.rename({'question_text': 'question'}, axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:41:47.766212Z","iopub.execute_input":"2021-05-23T01:41:47.766629Z","iopub.status.idle":"2021-05-23T01:41:47.934752Z","shell.execute_reply.started":"2021-05-23T01:41:47.766595Z","shell.execute_reply":"2021-05-23T01:41:47.933544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:17.553001Z","iopub.execute_input":"2021-05-23T01:17:17.553339Z","iopub.status.idle":"2021-05-23T01:17:17.560599Z","shell.execute_reply.started":"2021-05-23T01:17:17.553308Z","shell.execute_reply":"2021-05-23T01:17:17.559455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tập huấn luyện gồm 3 cột: ***qid***, ***question*** và ***target***.","metadata":{}},{"cell_type":"code","source":"# Đếm số lượng hàng dữ liệu của mỗi nhãn.\ntrain_df.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:43:13.972972Z","iopub.execute_input":"2021-05-23T01:43:13.973623Z","iopub.status.idle":"2021-05-23T01:43:13.995101Z","shell.execute_reply.started":"2021-05-23T01:43:13.973569Z","shell.execute_reply":"2021-05-23T01:43:13.993758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.ticker as ticker\n\nncount = train_df.shape[0]\n\nplt.figure(figsize=(7, 5))\n\nax = sns.countplot(data=train_df, x='target')\nplt.title('Distribution of Questions')\nplt.xlabel('Number of Axles')\n\n# Make twin axis\nax2=ax.twinx()\n\n# Switch so count axis is on right, frequency on left\nax2.yaxis.tick_left()\nax.yaxis.tick_right()\n\n# Also switch the labels over\nax.yaxis.set_label_position('right')\nax2.yaxis.set_label_position('left')\n\nax2.set_ylabel('Frequency [%]')\n\nfor p in ax.patches:\n    x=p.get_bbox().get_points()[:,0]\n    y=p.get_bbox().get_points()[1,1]\n    ax.annotate('{:.1f}%'.format(100.*y/ncount), (x.mean(), y), \n            ha='center', va='bottom') # set the alignment of the text\n\n# Use a LinearLocator to ensure the correct number of ticks\nax.yaxis.set_major_locator(ticker.LinearLocator(11))\n\n# Fix the frequency range to 0-100\nax2.set_ylim(0,100)\nax.set_ylim(0,ncount)\n\n# And use a MultipleLocator to ensure a tick spacing of 10\nax2.yaxis.set_major_locator(ticker.MultipleLocator(10))\n\n# Need to turn the grid on ax2 off, otherwise the gridlines end up on top of the bars\nax2.grid(None)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:17.589467Z","iopub.execute_input":"2021-05-23T01:17:17.589797Z","iopub.status.idle":"2021-05-23T01:17:17.995095Z","shell.execute_reply.started":"2021-05-23T01:17:17.589765Z","shell.execute_reply":"2021-05-23T01:17:17.99231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Từ các số liệu trên, tập dữ liệu bao gồm 1225312 câu hỏi chất lượng, chiếm 93,8%. Trong khi đó, có tổng cộng 80810 câu hỏi kém chất lượng, chiếm 6.2% tập dữ liệu. Như vậy, tập dữ liệu của chúng ta rất không cân bằng. ","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\ninsincere_qes = train_df[train_df['target'] == 1]\nprint(insincere_qes[:5].question)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:17.996603Z","iopub.execute_input":"2021-05-23T01:17:17.99698Z","iopub.status.idle":"2021-05-23T01:17:18.046225Z","shell.execute_reply.started":"2021-05-23T01:17:17.996942Z","shell.execute_reply":"2021-05-23T01:17:18.04473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\nsincere_qes = train_df[train_df['target'] == 0]\nprint(sincere_qes[-5:].question)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.047842Z","iopub.execute_input":"2021-05-23T01:17:18.04828Z","iopub.status.idle":"2021-05-23T01:17:18.143576Z","shell.execute_reply.started":"2021-05-23T01:17:18.048222Z","shell.execute_reply":"2021-05-23T01:17:18.142448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Words cloud","metadata":{}},{"cell_type":"code","source":"from wordcloud import WordCloud","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:23:53.639713Z","iopub.execute_input":"2021-05-23T01:23:53.640135Z","iopub.status.idle":"2021-05-23T01:23:53.760994Z","shell.execute_reply.started":"2021-05-23T01:23:53.640099Z","shell.execute_reply":"2021-05-23T01:23:53.760022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Word cloud của loại câu hỏi chất lượng.')\ninsincere_wordcloud = WordCloud(width=800, height=400, background_color ='black', min_font_size = 10).generate(str(train_df[train_df[\"target\"] == 1][\"question\"]))\nplt.figure(figsize=(15,6), facecolor=None)\nplt.imshow(insincere_wordcloud)\nplt.axis(\"off\")\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:24:08.237148Z","iopub.execute_input":"2021-05-23T01:24:08.237582Z","iopub.status.idle":"2021-05-23T01:24:08.882576Z","shell.execute_reply.started":"2021-05-23T01:24:08.237547Z","shell.execute_reply":"2021-05-23T01:24:08.88111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Word cloud của loại câu hỏi kém chất lượng.')\ninsincere_wordcloud = WordCloud(width=800, height=400, background_color ='black', min_font_size = 10).generate(str(train_df[train_df[\"target\"] == 0][\"question\"]))\nplt.figure(figsize=(15,6), facecolor=None)\nplt.imshow(insincere_wordcloud)\nplt.axis(\"off\")\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:24:29.179508Z","iopub.execute_input":"2021-05-23T01:24:29.180018Z","iopub.status.idle":"2021-05-23T01:24:29.817456Z","shell.execute_reply.started":"2021-05-23T01:24:29.179986Z","shell.execute_reply":"2021-05-23T01:24:29.816373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature engineering","metadata":{}},{"cell_type":"code","source":"import string\nimport re\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.stem import SnowballStemmer\n\nstop_words = stopwords.words('english')\nstop_words.remove('not')\nlemmatizer = WordNetLemmatizer()\nstemmer = SnowballStemmer('english')","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:28:44.065567Z","iopub.execute_input":"2021-05-23T01:28:44.065998Z","iopub.status.idle":"2021-05-23T01:28:44.073586Z","shell.execute_reply.started":"2021-05-23T01:28:44.065955Z","shell.execute_reply":"2021-05-23T01:28:44.072151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"contraction_dict = {\"dont\": \"do not\", \"aint\": \"is not\", \"isnt\": \"is not\", \"doesnt\": \"does not\", \"cant\": \"cannot\", \"mustnt\": \"must not\", \"hasnt\": \"has not\", \"havent\": \"have not\", \"arent\": \"are not\", \"ain't\": \"is not\", \"aren't\": \"are not\",\"can't\": \"cannot\", \"‘cause\": \"because\", \"could've\": \"could have\", \"couldn't\": \"could not\", \"didn't\": \"did not\", \"doesn't\": \"does not\", \"don't\": \"do not\", \"hadn't\": \"had not\", \"hasn't\": \"has not\", \"haven't\": \"have not\", \"he'd\": \"he would\",\"he'll\": \"he will\", \"he's\": \"he is\", \"how'd\": \"how did\", \"how'd'y\": \"how do you\", \"how'll\": \"how will\", \"how's\": \"how is\", \"I'd\": \"I would\", \"I'd've\": \"I would have\", \"I'll\": \"I will\", \"I'll've\": \"I will have\",\"I'm\": \"I am\", \"Iam\": \"I am\", \"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \"i'll\": \"i will\", \"i'll've\": \"i will have\",\"i'm\": \"i am\", \"i've\": \"i have\", \"isn't\": \"is not\", \"it'd\": \"it would\", \"it'd've\": \"it would have\", \"it'll\": \"it will\", \"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\", \"ma'am\": \"madam\", \"mayn't\": \"may not\", \"might've\": \"might have\",\"mightn't\": \"might not\",\"mightn't've\": \"might not have\", \"must've\": \"must have\", \"mustn't\": \"must not\", \"mustn't've\": \"must not have\", \"needn't\": \"need not\", \"needn't've\": \"need not have\",\"o'clock\": \"of the clock\", \"oughtn't\": \"ought not\", \"oughtn't've\": \"ought not have\", \"shan't\": \"shall not\", \"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\", \"she'd\": \"she would\", \"she'd've\": \"she would have\", \"she'll\": \"she will\", \"she'll've\": \"she will have\", \"she's\": \"she is\", \"should've\": \"should have\", \"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \"that's\": \"that is\", \"there'd\": \"there would\", \"there'd've\": \"there would have\", \"there's\": \"there is\", \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\", \"they'll\": \"they will\", \"they'll've\": \"they will have\", \"they're\": \"they are\", \"they've\": \"they have\", \"to've\": \"to have\", \"wasn't\": \"was not\", \"we'd\": \"we would\", \"we'd've\": \"we would have\", \"we'll\": \"we will\", \"we'll've\": \"we will have\", \"we're\": \"we are\", \"we've\": \"we have\", \"weren't\": \"were not\", \"what'll\": \"what will\", \"what'll've\": \"what will have\", \"what're\": \"what are\", \"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\", \"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\", \"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\", \"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\", \"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \"you're\": \"you are\", \"you've\": \"you have\"}\n\n# if contraction_dict.has_key('aya'):\n#     print('yaya')\npunctuation = string.punctuation ","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:28:56.073846Z","iopub.execute_input":"2021-05-23T01:28:56.074186Z","iopub.status.idle":"2021-05-23T01:28:56.089065Z","shell.execute_reply.started":"2021-05-23T01:28:56.074158Z","shell.execute_reply":"2021-05-23T01:28:56.088332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def replace_contractions(question):\n    return [contraction_dict.get(token) if (contraction_dict.get(token) != None) else token for token in question]","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:28:58.589357Z","iopub.execute_input":"2021-05-23T01:28:58.5897Z","iopub.status.idle":"2021-05-23T01:28:58.595483Z","shell.execute_reply.started":"2021-05-23T01:28:58.589669Z","shell.execute_reply":"2021-05-23T01:28:58.593973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Feature Engineering on train_df Data\ndef create_features(df_):\n    \"\"\"Retrieve from the text column the nb of : words, unique words, characters, stopwords,\n    punctuations, upper/lower case char, title...\"\"\"\n    \n    df_[\"question_length\"] = df_[\"question\"].apply(lambda x: len(x))\n    df_[\"nb_words\"] = df_[\"question\"].apply(lambda x: len(x.split()))\n    df_[\"nb_unique_words\"] = df_[\"question\"].apply(lambda x: len(set(str(x).split())))\n    df_[\"nb_chars\"] = df_[\"question\"].apply(lambda x: len(str(x)))\n    df_[\"nb_stopwords\"] = df_[\"question\"].apply(lambda x : len([nw for nw in str(x).split() if nw.lower() in stop_words]))\n    df_[\"nb_punctuation\"] = df_[\"question\"].apply(lambda x : len([np for np in str(x) if np in punctuation]))\n    df_[\"nb_uppercase\"] = df_[\"question\"].apply(lambda x : len([nu for nu in str(x).split() if nu.isupper()]))\n    df_[\"nb_lowercase\"] = df_[\"question\"].apply(lambda x : len([nl for nl in str(x).split() if nl.islower()]))\n    df_[\"nb_title\"] = df_[\"question\"].apply(lambda x : len([nl for nl in str(x).split() if nl.istitle()]))\n    return df_","metadata":{"execution":{"iopub.status.busy":"2021-05-23T05:05:14.539169Z","iopub.execute_input":"2021-05-23T05:05:14.539599Z","iopub.status.idle":"2021-05-23T05:05:14.551927Z","shell.execute_reply.started":"2021-05-23T05:05:14.539498Z","shell.execute_reply":"2021-05-23T05:05:14.550766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = create_features(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:29:04.778305Z","iopub.execute_input":"2021-05-23T01:29:04.778692Z","iopub.status.idle":"2021-05-23T01:29:56.71795Z","shell.execute_reply.started":"2021-05-23T01:29:04.778662Z","shell.execute_reply":"2021-05-23T01:29:56.716946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.773263Z","iopub.execute_input":"2021-05-23T01:17:18.773565Z","iopub.status.idle":"2021-05-23T01:17:18.786854Z","shell.execute_reply.started":"2021-05-23T01:17:18.773537Z","shell.execute_reply":"2021-05-23T01:17:18.785175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Số liệu thống kê của loại câu hỏi chất lượng.**","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.float_format', lambda x: '%.3f' % x)\ntrain_df[train_df['target'] == 0].describe()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:29:56.719783Z","iopub.execute_input":"2021-05-23T01:29:56.720415Z","iopub.status.idle":"2021-05-23T01:29:57.987159Z","shell.execute_reply.started":"2021-05-23T01:29:56.720368Z","shell.execute_reply":"2021-05-23T01:29:57.985615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Số liệu thống kê của loại câu hỏi kém chất lượng.**","metadata":{}},{"cell_type":"code","source":"pd.set_option('display.float_format', lambda x: '%.3f' % x)\ntrain_df[train_df['target'] == 1].describe()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:30:11.972724Z","iopub.execute_input":"2021-05-23T01:30:11.973111Z","iopub.status.idle":"2021-05-23T01:30:12.103004Z","shell.execute_reply.started":"2021-05-23T01:30:11.973078Z","shell.execute_reply":"2021-05-23T01:30:12.101711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_feat = ['question_length', 'nb_unique_words', 'nb_chars', 'nb_stopwords', \\\n            'nb_punctuation', 'nb_uppercase', 'nb_lowercase', 'nb_title', 'target'] \n\ndf_sample = train_df[num_feat].sample(n=round(train_df.shape[0]/6), random_state=42)\n\nplt.figure(figsize=(16,10))\nsns.pairplot(data=df_sample, hue='target')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:30:27.748987Z","iopub.execute_input":"2021-05-23T01:30:27.749368Z","iopub.status.idle":"2021-05-23T01:41:42.507782Z","shell.execute_reply.started":"2021-05-23T01:30:27.749335Z","shell.execute_reply":"2021-05-23T01:41:42.505873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Qua hai bảng số liệu, ta có thể thấy độ dài trung bình của loại câu hỏi kém chất lượng ngắn hơn loại câu hỏi chất lượng. Các features khác cũng tương tự như vậy, có thể là do chúng chịu ảnh hưởng của sự chênh lệch độ dài trên. ","metadata":{}},{"cell_type":"code","source":"mask = np.zeros_like(train_df[num_feat].corr(), dtype=np.bool) \nmask[np.triu_indices_from(mask)] = True \n\nf, ax = plt.subplots(figsize=(10, 10))\nplt.title('Question features Correlation Matrix',fontsize=20)\n\nsns.heatmap(train_df[num_feat].corr(),square=True, linewidths=0.25,vmax=0.7,cmap=\"YlGnBu\",\n            linecolor='w',annot=True,annot_kws={\"size\":10},mask=mask,cbar_kws={\"shrink\": .9});","metadata":{"execution":{"iopub.status.busy":"2021-05-23T02:40:35.908656Z","iopub.execute_input":"2021-05-23T02:40:35.909266Z","iopub.status.idle":"2021-05-23T02:40:35.930615Z","shell.execute_reply.started":"2021-05-23T02:40:35.90922Z","shell.execute_reply":"2021-05-23T02:40:35.928957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 5))\n\ntarget_correlation = train_df.corr()['target'][1:]\nplt.plot(target_correlation)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:41:45.083016Z","iopub.execute_input":"2021-05-23T01:41:45.083314Z","iopub.status.idle":"2021-05-23T01:41:46.536283Z","shell.execute_reply.started":"2021-05-23T01:41:45.083285Z","shell.execute_reply":"2021-05-23T01:41:46.535259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy, độ dài của một câu hỏi phân biệt loại của nó tốt nhất, tuy nhiên, tỉ lệ này là khá ít (chỉ 0.18). Tóm lại, mọi features đều có sự liên quan nhất định tới phân loại của câu hỏi, mặc dù sự liên quan này là không cao.","metadata":{}},{"cell_type":"markdown","source":"# Data preprocessing","metadata":{}},{"cell_type":"code","source":"def qes_preprocessing(qes):\n    # Data cleaning:\n    qes = re.sub(re.compile('<.*?>'), '', qes)\n    qes = re.sub('[^A-Za-z0-9]+', ' ', qes)\n\n    # Lowercase:\n    qes = qes.lower()\n\n    # Tokenization:\n    tokens = word_tokenize(qes)\n\n    # Contractions replacement:\n    tokens = [contraction_dict.get(token) if (contraction_dict.get(token) != None) else token for token in tokens]\n\n    # Stop words removal:\n    tokens = [w for w in tokens if w not in stop_words]\n\n    # Stemming:\n    tokens = [stemmer.stem(token) for token in tokens]\n    \n    # Lemmatization:\n    tokens = [lemmatizer.lemmatize(w) for w in tokens]\n\n    # Join words after preprocessed:\n    qes = ' '.join(tokens) \n\n    return qes","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.918452Z","iopub.status.idle":"2021-05-23T01:17:18.918903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tqdm.pandas()\ntrain_df['preprocessed_questions'] = train_df['question'].progress_apply(qes_preprocessing)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.920366Z","iopub.status.idle":"2021-05-23T01:17:18.920821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train model on training set","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.linear_model import LogisticRegression\n\npipeline = Pipeline([(\"cv\", CountVectorizer(analyzer=\"word\", ngram_range=(1,4), max_df=0.9)),\n                     (\"clf\", LogisticRegression(solver=\"saga\", class_weight=\"balanced\", C=0.45, max_iter=300, verbose=1, n_jobs=-1))])\n\nX = train_df['preprocessed_questions']\ny = train_df['target']","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.921857Z","iopub.status.idle":"2021-05-23T01:17:18.922363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_model = pipeline.fit(X, y)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.923363Z","iopub.status.idle":"2021-05-23T01:17:18.923801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load test data and make prediction","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/quora-insincere-questions-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.927521Z","iopub.status.idle":"2021-05-23T01:17:18.927961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.928835Z","iopub.status.idle":"2021-05-23T01:17:18.929289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['preprocessed'] = test_df['question_text'].apply(qes_preprocessing)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.930145Z","iopub.status.idle":"2021-05-23T01:17:18.930624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = lr_model.predict(test_df['preprocessed'])","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.931619Z","iopub.status.idle":"2021-05-23T01:17:18.932057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['prediction'] = predictions\nresults = test_df[['qid', 'prediction']]\nresults.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-05-23T01:17:18.934581Z","iopub.status.idle":"2021-05-23T01:17:18.935022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.head()","metadata":{},"execution_count":null,"outputs":[]}]}