{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\nimport string\n\nfrom nltk.corpus import stopwords\nfrom nltk.stem.snowball import SnowballStemmer\nfrom sklearn.manifold import TSNE\nimport re","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-08T13:43:27.878713Z","iopub.execute_input":"2022-01-08T13:43:27.879111Z","iopub.status.idle":"2022-01-08T13:43:34.905481Z","shell.execute_reply.started":"2022-01-08T13:43:27.878992Z","shell.execute_reply":"2022-01-08T13:43:34.904233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**1. Phân tích dữ liệu**","metadata":{}},{"cell_type":"code","source":"train_data_path = \"/kaggle/input/quora-insincere-questions-classification/train.csv\"\ntest_data_path = \"/kaggle/input/quora-insincere-questions-classification/test.csv\"\ntrain_data = pd.read_csv(train_data_path)\ntest_data = pd.read_csv(test_data_path)\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:43:34.907758Z","iopub.execute_input":"2022-01-08T13:43:34.908075Z","iopub.status.idle":"2022-01-08T13:43:41.673419Z","shell.execute_reply.started":"2022-01-08T13:43:34.908037Z","shell.execute_reply":"2022-01-08T13:43:41.672531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:43:41.674612Z","iopub.execute_input":"2022-01-08T13:43:41.674842Z","iopub.status.idle":"2022-01-08T13:43:41.687128Z","shell.execute_reply.started":"2022-01-08T13:43:41.674814Z","shell.execute_reply":"2022-01-08T13:43:41.686323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\ndf = pd.read_csv(train_data_path)\ndf['target'].value_counts().plot.bar(title='Target')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:43:41.689015Z","iopub.execute_input":"2022-01-08T13:43:41.689366Z","iopub.status.idle":"2022-01-08T13:43:45.405264Z","shell.execute_reply.started":"2022-01-08T13:43:41.689331Z","shell.execute_reply":"2022-01-08T13:43:45.404187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"value_counts = train_data['target'].value_counts()\nvalue_counts_percentage = train_data['target'].value_counts(normalize=True).mul(100).round(1).astype(str) + '%'\npd.concat([value_counts, value_counts_percentage], axis=1, keys=['Counts', 'Percentage'])","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:43:45.406818Z","iopub.execute_input":"2022-01-08T13:43:45.407064Z","iopub.status.idle":"2022-01-08T13:43:45.436616Z","shell.execute_reply.started":"2022-01-08T13:43:45.407031Z","shell.execute_reply":"2022-01-08T13:43:45.435317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Biểu đồ Histogram tần suất của từ\nword_length_list = [len(x.split()) for x in train_data['question_text'] if len(x.split()) < 80]\nchar_length_list = [len(x) for x in train_data['question_text'] if len(x) < 200]\nfig, axs = plt.subplots(1, 2, sharey=True, tight_layout=True)\naxs[0].hist(word_length_list, bins=25)\naxs[0].set_title('Words in Questions')\n\naxs[1].hist(char_length_list, bins=25)\naxs[1].set_title('Length of Questions')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:43:45.438107Z","iopub.execute_input":"2022-01-08T13:43:45.438526Z","iopub.status.idle":"2022-01-08T13:44:03.950671Z","shell.execute_reply.started":"2022-01-08T13:43:45.438482Z","shell.execute_reply":"2022-01-08T13:44:03.949568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**2. Xử lý dữ liệu**","metadata":{}},{"cell_type":"code","source":"from nltk.tokenize import word_tokenize\nstop_words = list(stopwords.words('english'))\n\npuncts = [',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', '•',  '~', '@', '£', \n '·', '_', '{', '}', '©', '^', '®', '`',  '<', '→', '°', '€', '™', '›',  '♥', '←', '×', '§', '″', '′', 'Â', '█', '½', 'à', '…', \n '“', '★', '”', '–', '●', 'â', '►', '−', '¢', '²', '¬', '░', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', '—', '‹', '─', \n '▒', '：', '¼', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', 'é', '¯', '♦', '¤', '▲', 'è', '¸', '¾', 'Ã', '⋅', '‘', '∞', \n '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '³', '・', '╦', '╣', '╔', '╗', '▬', '❤', 'ï', 'Ø', '¹', '≤', '‡', '√', ]\n# Xoá các dấu câu\ndef clean_text(x):\n    x = str(x)\n    for punct in puncts:\n        x = x.replace(punct, f' {punct} ')\n    return x\n\n# Xoá chữ số\ndef clean_numbers(x):\n    return re.sub('[0-9]{2}', ' ', x)\n\n#Xoá stop_words\ndef remove_stopwords(x):\n    word_token = word_tokenize(x)\n    filtered = [w for w in word_token if not w in stop_words]\n    x = \" \".join(filtered)\n    return x","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:44:03.952227Z","iopub.execute_input":"2022-01-08T13:44:03.952490Z","iopub.status.idle":"2022-01-08T13:44:03.972106Z","shell.execute_reply.started":"2022-01-08T13:44:03.952455Z","shell.execute_reply":"2022-01-08T13:44:03.971103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_clean(x):\n  x = clean_text(x)\n  x = clean_numbers(x)\n  x = remove_stopwords(x)\n  return x","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:44:03.973462Z","iopub.execute_input":"2022-01-08T13:44:03.973740Z","iopub.status.idle":"2022-01-08T13:44:03.982259Z","shell.execute_reply.started":"2022-01-08T13:44:03.973709Z","shell.execute_reply":"2022-01-08T13:44:03.981419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#clean dữ liệu tập train và test\ntrain_data['question_text_cleaned'] = train_data['question_text'].apply(lambda x: data_clean(x))\ntest_data['question_text_cleaned'] = test_data['question_text'].apply(lambda x: data_clean(x))\ndisplay(train_data.head(),test_data.head())","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:44:03.983632Z","iopub.execute_input":"2022-01-08T13:44:03.983864Z","iopub.status.idle":"2022-01-08T13:52:10.551962Z","shell.execute_reply.started":"2022-01-08T13:44:03.983837Z","shell.execute_reply":"2022-01-08T13:52:10.551106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train_data.head(),test_data.head())","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:52:10.555485Z","iopub.execute_input":"2022-01-08T13:52:10.555832Z","iopub.status.idle":"2022-01-08T13:52:10.575111Z","shell.execute_reply.started":"2022-01-08T13:52:10.555789Z","shell.execute_reply":"2022-01-08T13:52:10.574226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.svm import LinearSVC\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import classification_report\nfrom sklearn.feature_extraction.text import TfidfVectorizer","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:52:10.576861Z","iopub.execute_input":"2022-01-08T13:52:10.578775Z","iopub.status.idle":"2022-01-08T13:52:10.584073Z","shell.execute_reply.started":"2022-01-08T13:52:10.578732Z","shell.execute_reply":"2022-01-08T13:52:10.583220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**3. Vector hoá dữ liệu**","metadata":{}},{"cell_type":"code","source":"tfidf = TfidfVectorizer(ngram_range=(1, 3))\n\ndef predict_linearSVC(X_train,y_train,X_test):\n    tfidf.fit(X_train)\n    X_train = tfidf.transform(X_train)\n    X_test = tfidf.transform(X_test)\n    svm = LinearSVC()\n    svm.fit(X_train,y_train)\n    return svm.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:52:10.586045Z","iopub.execute_input":"2022-01-08T13:52:10.586718Z","iopub.status.idle":"2022-01-08T13:52:10.597160Z","shell.execute_reply.started":"2022-01-08T13:52:10.586673Z","shell.execute_reply":"2022-01-08T13:52:10.596133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**4. Mô hình**","metadata":{}},{"cell_type":"code","source":"# Phân chia dữ liệu test = 20%, dữ liệu train = 80%\nX = train_data.question_text_cleaned\ny = train_data.target\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:57:04.532435Z","iopub.execute_input":"2022-01-08T13:57:04.532752Z","iopub.status.idle":"2022-01-08T13:57:04.968662Z","shell.execute_reply.started":"2022-01-08T13:57:04.532720Z","shell.execute_reply":"2022-01-08T13:57:04.967680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict = predict_linearSVC(X_train,y_train,X_test)\nprint('F1 score :', f1_score(predict, y_test), '\\n')\nprint(classification_report(y_test, predict))","metadata":{"execution":{"iopub.status.busy":"2022-01-08T13:52:11.028323Z","iopub.execute_input":"2022-01-08T13:52:11.028630Z","iopub.status.idle":"2022-01-08T13:55:27.313195Z","shell.execute_reply.started":"2022-01-08T13:52:11.028596Z","shell.execute_reply":"2022-01-08T13:55:27.312314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**5. Tạo submission**","metadata":{}},{"cell_type":"code","source":"pred_data = test_data['question_text_cleaned']\n\npredict = predict_linearSVC(X_train,y_train,pred_data)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:08:10.720151Z","iopub.execute_input":"2022-01-08T15:08:10.720545Z","iopub.status.idle":"2022-01-08T15:11:22.008864Z","shell.execute_reply.started":"2022-01-08T15:08:10.720510Z","shell.execute_reply":"2022-01-08T15:11:22.007764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.DataFrame({'qid':test_data['qid'].values})\nsubmit['prediction'] = predict\nsubmit.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:12:06.091787Z","iopub.execute_input":"2022-01-08T15:12:06.092112Z","iopub.status.idle":"2022-01-08T15:12:07.027525Z","shell.execute_reply.started":"2022-01-08T15:12:06.092079Z","shell.execute_reply":"2022-01-08T15:12:07.026803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit","metadata":{"execution":{"iopub.status.busy":"2022-01-08T15:17:01.674308Z","iopub.execute_input":"2022-01-08T15:17:01.674650Z","iopub.status.idle":"2022-01-08T15:17:01.689784Z","shell.execute_reply.started":"2022-01-08T15:17:01.674618Z","shell.execute_reply":"2022-01-08T15:17:01.689100Z"},"trusted":true},"execution_count":null,"outputs":[]}]}