{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"19021374 - Pham Thi Minh Trang","metadata":{}},{"cell_type":"markdown","source":"# Mô tả bài toán\nPhân loại các câu hỏi thật (sincere) và câu hỏi chỉ mang tính câu fame (insincere).\n\n**Input**: Các câu hỏi ở dạng text\n\n**Output**: 0 nếu câu hỏi là sincere và 1 nếu là insincere.","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport re\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:41:57.212088Z","iopub.execute_input":"2022-01-08T22:41:57.212913Z","iopub.status.idle":"2022-01-08T22:41:57.223725Z","shell.execute_reply.started":"2022-01-08T22:41:57.212869Z","shell.execute_reply":"2022-01-08T22:41:57.22281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Phân tích dữ liệu\nDữ liệu được cung cấp sẵn gồm có hai file csv để lưu dữ liệu train và test, một file zip chứa 4 loại pretrained word embeddings.\n","metadata":{}},{"cell_type":"code","source":"train_path = '../input/quora-insincere-questions-classification/train.csv'\ntest_path = '../input/quora-insincere-questions-classification/test.csv'\nemb_path = '../input/quora-insincere-questions-classification/embeddings.zip'","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:41:59.804583Z","iopub.execute_input":"2022-01-08T22:41:59.804883Z","iopub.status.idle":"2022-01-08T22:41:59.809149Z","shell.execute_reply.started":"2022-01-08T22:41:59.804854Z","shell.execute_reply":"2022-01-08T22:41:59.808226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(train_path)\ntest_data = pd.read_csv(test_path)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:42:02.483128Z","iopub.execute_input":"2022-01-08T22:42:02.483447Z","iopub.status.idle":"2022-01-08T22:42:09.049751Z","shell.execute_reply.started":"2022-01-08T22:42:02.483412Z","shell.execute_reply":"2022-01-08T22:42:09.048804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dữ liệu huấn luyện gồm có ba trường: \n- qid: đánh index định danh cho câu hỏi. Không ảnh hưởng gì đến việc huấn luyện.\n- question_text: nội dung câu hỏi cần phải phân loại.\n- target: phân loại của câu hỏi trên. Giá trị 0 là sincere, 1 là insincere.","metadata":{}},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:12:15.597606Z","iopub.execute_input":"2022-01-08T22:12:15.597974Z","iopub.status.idle":"2022-01-08T22:12:15.634013Z","shell.execute_reply.started":"2022-01-08T22:12:15.597929Z","shell.execute_reply":"2022-01-08T22:12:15.633416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data exploring metrics:\n- number of samples\n- number of classes\n- number of samples per class\n- number of words per sample\n- frequency distribution of words\n- distribution of samples\n\nCalculate the number of samples/number of words per sample ratio.\n","metadata":{}},{"cell_type":"code","source":"num_sample = len(train_data)\nprint('Number of samples: ', num_sample)\n# print('Number of classes', max(train_data['target'])+1)\nnum_word = [len(s.split()) for s in train_data['question_text']]\nnum_wps = np.median(num_word)\nprint('Number of words per sample: ', num_wps)\nnum_sample/num_wps","metadata":{"execution":{"iopub.status.busy":"2022-01-08T19:16:51.637146Z","iopub.execute_input":"2022-01-08T19:16:51.638048Z","iopub.status.idle":"2022-01-08T19:16:53.268629Z","shell.execute_reply.started":"2022-01-08T19:16:51.638002Z","shell.execute_reply":"2022-01-08T19:16:53.267996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-01-08T19:16:53.269604Z","iopub.execute_input":"2022-01-08T19:16:53.270239Z","iopub.status.idle":"2022-01-08T19:16:53.287291Z","shell.execute_reply.started":"2022-01-08T19:16:53.270204Z","shell.execute_reply":"2022-01-08T19:16:53.286661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Imbalanced data: 1225312-80810","metadata":{}},{"cell_type":"code","source":"#cắt bớt data để 2 nhóm cân bằng\nfrom random import sample\ntrain_data_balance = train_data[train_data['target'] == 1]\ntrain_data_balance = train_data_balance.append(train_data[train_data['target'] == 0].sample(len(train_data[train_data['target']==1])))\n# shuffle all rows and reset index columns\ntrain_data_balance = train_data_balance.sample(frac=1).reset_index(drop=True)\ntrain_data_balance","metadata":{"execution":{"iopub.status.busy":"2022-01-08T19:16:53.288325Z","iopub.execute_input":"2022-01-08T19:16:53.289008Z","iopub.status.idle":"2022-01-08T19:16:53.594284Z","shell.execute_reply.started":"2022-01-08T19:16:53.288953Z","shell.execute_reply":"2022-01-08T19:16:53.593723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Clean data\n- xóa ngắt câu và kí tự lạ: done\n- xóa số: done\n- xóa biểu thức, links\n- sửa sai chính tả: done\n- viết thường: done\n- xóa từ hiếm","metadata":{}},{"cell_type":"code","source":"punctuation_list =[',', '.', '\"', ':', ')', '(', '-', '!', '?', '|', ';', \"'\", '$', '&', '/', '[', ']', '>', '%', '=', '#', '*', '+', '\\\\', \n        '•', '~', '@', '£', '·', '_', '{', '}', '©', '^', '®', '`', '<', '→', '°', '€', '™', '›', '♥', '←', '×', '§', '″', '′', \n        '█', '…', '“', '★', '”', '–', '●', '►', '−', '¢', '¬', '░', '¡', '¶', '↑', '±', '¿', '▾', '═', '¦', '║', '―', '¥', '▓', \n        '—', '‹', '─', '▒', '：', '⊕', '▼', '▪', '†', '■', '’', '▀', '¨', '▄', '♫', '☆', '¯', '♦', '¤', '▲', '¸', '⋅', '‘', '∞', \n        '∙', '）', '↓', '、', '│', '（', '»', '，', '♪', '╩', '╚', '・', '╦', '╣', '╔', '╗', '▬', '❤', '≤', '‡', '√', '◄', '━', \n        '⇒', '▶', '≥', '╝', '♡', '◊', '。', '✈', '≡', '☺', '✔', '↵', '≈', '✓', '♣', '☎', '℃', '◦', '└', '‟', '～', '！', '○', \n        '◆', '№', '♠', '▌', '✿', '▸', '⁄', '□', '❖', '✦', '．', '÷', '｜', '┃', '／', '￥', '╠', '↩', '✭', '▐', '☼', '☻', '┐', \n        '├', '«', '∼', '┌', '℉', '☮', '฿', '≦', '♬', '✧', '〉', '－', '⌂', '✖', '･', '◕', '※', '‖', '◀', '‰', '\\x97', '↺', \n        '∆', '┘', '┬', '╬', '،', '⌘', '⊂', '＞', '〈', '⎙', '？', '☠', '⇐', '▫', '∗', '∈', '≠', '♀', '♔', '˚', '℗', '┗', '＊', \n        '┼', '❀', '＆', '∩', '♂', '‿', '∑', '‣', '➜', '┛', '⇓', '☯', '⊖', '☀', '┳', '；', '∇', '⇑', '✰', '◇', '♯', '☞', '´', \n        '↔', '┏', '｡', '◘', '∂', '✌', '♭', '┣', '┴', '┓', '✨', '\\xa0', '˜', '❥', '┫', '℠', '✒', '［', '∫', '\\x93', '≧', '］', \n        '\\x94', '∀', '♛', '\\x96', '∨', '◎', '↻', '⇩', '＜', '≫', '✩', '✪', '♕', '؟', '₤', '☛', '╮', '␊', '＋', '┈', '％', \n        '╋', '▽', '⇨', '┻', '⊗', '￡', '।', '▂', '✯', '▇', '＿', '➤', '✞', '＝', '▷', '△', '◙', '▅', '✝', '∧', '␉', '☭', \n        '┊', '╯', '☾', '➔', '∴', '\\x92', '▃', '↳', '＾', '׳', '➢', '╭', '➡', '＠', '⊙', '☢', '˝', '∏', '„', '∥', '❝', '☐', \n        '▆', '╱', '⋙', '๏', '☁', '⇔', '▔', '\\x91', '➚', '◡', '╰', '\\x85', '♢', '˙', '۞', '✘', '✮', '☑', '⋆', 'ⓘ', '❒', \n        '☣', '✉', '⌊', '➠', '∣', '❑', '◢', 'ⓒ', '\\x80', '〒', '∕', '▮', '⦿', '✫', '✚', '⋯', '♩', '☂', '❞', '‗', '܂', '☜', \n        '‾', '✜', '╲', '∘', '⟩', '＼', '⟨', '·', '✗', '♚', '∅', 'ⓔ', '◣', '͡', '‛', '❦', '◠', '✄', '❄', '∃', '␣', '≪', '｢', \n        '≅', '◯', '☽', '∎', '｣', '❧', '̅', 'ⓐ', '↘', '⚓', '▣', '˘', '∪', '⇢', '✍', '⊥', '＃', '⎯', '↠', '۩', '☰', '◥', \n        '⊆', '✽', '⚡', '↪', '❁', '☹', '◼', '☃', '◤', '❏', 'ⓢ', '⊱', '➝', '̣', '✡', '∠', '｀', '▴', '┤', '∝', '♏', 'ⓐ', \n        '✎', ';', '␤', '＇', '❣', '✂', '✤', 'ⓞ', '☪', '✴', '⌒', '˛', '♒', '＄', '✶', '▻', 'ⓔ', '◌', '◈', '❚', '❂', '￦', \n        '◉', '╜', '̃', '✱', '╖', '❉', 'ⓡ', '↗', 'ⓣ', '♻', '➽', '׀', '✲', '✬', '☉', '▉', '≒', '☥', '⌐', '♨', '✕', 'ⓝ', \n        '⊰', '❘', '＂', '⇧', '̵', '➪', '▁', '▏', '⊃', 'ⓛ', '‚', '♰', '́', '✏', '⏑', '̶', 'ⓢ', '⩾', '￠', '❍', '≃', '⋰', '♋', \n        '､', '̂', '❋', '✳', 'ⓤ', '╤', '▕', '⌣', '✸', '℮', '⁺', '▨', '╨', 'ⓥ', '♈', '❃', '☝', '✻', '⊇', '≻', '♘', '♞', \n        '◂', '✟', '⌠', '✠', '☚', '✥', '❊', 'ⓒ', '⌈', '❅', 'ⓡ', '♧', 'ⓞ', '▭', '❱', 'ⓣ', '∟', '☕', '♺', '∵', '⍝', 'ⓑ', \n        '✵', '✣', '٭', '♆', 'ⓘ', '∶', '⚜', '◞', '்', '✹', '➥', '↕', '̳', '∷', '✋', '➧', '∋', '̿', 'ͧ', '┅', '⥤', '⬆', '⋱', \n        '☄', '↖', '⋮', '۔', '♌', 'ⓛ', '╕', '♓', '❯', '♍', '▋', '✺', '⭐', '✾', '♊', '➣', '▿', 'ⓑ', '♉', '⏠', '◾', '▹', \n        '⩽', '↦', '╥', '⍵', '⌋', '։', '➨', '∮', '⇥', 'ⓗ', 'ⓓ', '⁻', '⎝', '⌥', '⌉', '◔', '◑', '✼', '♎', '♐', '╪', '⊚', \n        '☒', '⇤', 'ⓜ', '⎠', '◐', '⚠', '╞', '◗', '⎕', 'ⓨ', '☟', 'ⓟ', '♟', '❈', '↬', 'ⓓ', '◻', '♮', '❙', '♤', '∉', '؛', \n        '⁂', 'ⓝ', '־', '♑', '╫', '╓', '╳', '⬅', '☔', '☸', '┄', '╧', '׃', '⎢', '❆', '⋄', '⚫', '̏', '☏', '➞', '͂', '␙', \n        'ⓤ', '◟', '̊', '⚐', '✙', '↙', '̾', '℘', '✷', '⍺', '❌', '⊢', '▵', '✅', 'ⓖ', '☨', '▰', '╡', 'ⓜ', '☤', '∽', '╘', \n        '˹', '↨', '♙', '⬇', '♱', '⌡', '⠀', '╛', '❕', '┉', 'ⓟ', '̀', '♖', 'ⓚ', '┆', '⎜', '◜', '⚾', '⤴', '✇', '╟', '⎛', \n        '☩', '➲', '➟', 'ⓥ', 'ⓗ', '⏝', '◃', '╢', '↯', '✆', '˃', '⍴', '❇', '⚽', '╒', '̸', '♜', '☓', '➳', '⇄', '☬', '⚑', \n        '✐', '⌃', '◅', '▢', '❐', '∊', '☈', '॥', '⎮', '▩', 'ு', '⊹', '‵', '␔', '☊', '➸', '̌', '☿', '⇉', '⊳', '╙', 'ⓦ', \n        '⇣', '｛', '̄', '↝', '⎟', '▍', '❗', '״', '΄', '▞', '◁', '⛄', '⇝', '⎪', '♁', '⇠', '☇', '✊', 'ி', '｝', '⭕', '➘', \n        '⁀', '☙', '❛', '❓', '⟲', '⇀', '≲', 'ⓕ', '⎥', '\\u06dd', 'ͤ', '₋', '̱', '̎', '♝', '≳', '▙', '➭', '܀', 'ⓖ', '⇛', '▊', \n        '⇗', '̷', '⇱', '℅', 'ⓧ', '⚛', '̐', '̕', '⇌', '␀', '≌', 'ⓦ', '⊤', '̓', '☦', 'ⓕ', '▜', '➙', 'ⓨ', '⌨', '◮', '☷', \n        '◍', 'ⓚ', '≔', '⏩', '⍳', '℞', '┋', '˻', '▚', '≺', 'ْ', '▟', '➻', '̪', '⏪', '̉', '⎞', '┇', '⍟', '⇪', '▎', '⇦', '␝', \n        '⤷', '≖', '⟶', '♗', '̴', '♄', 'ͨ', '̈', '❜', '̡', '▛', '✁', '➩', 'ா', '˂', '↥', '⏎', '⎷', '̲', '➖', '↲', '⩵', '̗', '❢', \n        '≎', '⚔', '⇇', '̑', '⊿', '̖', '☍', '➹', '⥊', '⁁', '✢']\n# chứa cả misspell và viết tắt và uk us\nmisspell_dict = {'colour': 'color', 'centre': 'center', 'favourite': 'favorite', 'travelling': 'traveling', 'counselling': 'counseling',\n                'theatre': 'theater', 'cancelled': 'canceled', 'labour': 'labor', 'organisation': 'organization', 'wwii': 'world war 2',\n                'citicise': 'criticize', 'youtu ': 'youtube ', 'Qoura': 'Quora', 'sallary': 'salary', 'Whta': 'What', 'narcisist': 'narcissist',\n                'howdo': 'how do', 'whatare': 'what are', 'howcan': 'how can', 'howmuch': 'how much', 'howmany': 'how many', 'whydo': 'why do', \n                'doI': 'do I', 'theBest': 'the best', 'howdoes': 'how does', 'mastrubation': 'masturbation', 'mastrubate': 'masturbate', \n                \"mastrubating\": 'masturbating', 'pennis': 'penis', 'Etherium': 'bitcoin', 'narcissit': 'narcissist', 'bigdata': 'big data', \n                '2k17': '2017', '2k18': '2018', 'qouta': 'quota', 'exboyfriend': 'ex boyfriend', 'airhostess': 'air hostess', \"whst\": 'what', \n                'watsapp': 'whatsapp', 'demonitisation': 'demonetization', 'demonitization': 'demonetization', 'demonetisation': 'demonetization', \n                'electroneum':'bitcoin','nanodegree':'degree','hotstar':'star','dream11':'dream','ftre':'fire','tensorflow':'framework',\n                'unocoin':'bitcoin','lnmiit':'limit','unacademy':'academy','altcoin':'bitcoin','altcoins':'bitcoin','litecoin':'bitcoin',\n                'coinbase':'bitcoin','cryptocurency':'cryptocurrency','simpliv':'simple','quoras':'quora','schizoids':'psychopath',\n                'remainers':'remainder','twinflame':'soulmate','quorans':'quora','brexit':'demonetized','iiest':'institute','dceu':'comics',\n                'pessat':'exam','uceed':'college','bhakts':'devotee','boruto':'anime','cryptocoin':'bitcoin','blockchains':'blockchain',\n                'fiancee':'fiance','redmi':'smartphone','oneplus':'smartphone','qoura':'quora','deepmind':'framework','ryzen':'cpu',\n                'whattsapp':'whatsapp','undertale':'adventure','zenfone':'smartphone','cryptocurencies':'cryptocurrencies','koinex':'bitcoin',\n                'zebpay':'bitcoin','binance':'bitcoin','whtsapp':'whatsapp','reactjs':'framework','bittrex':'bitcoin','bitconnect':'bitcoin',\n                'bitfinex':'bitcoin','yourquote':'your quote','whyis':'why is','jiophone':'smartphone','dogecoin':'bitcoin','onecoin':'bitcoin',\n                'poloniex':'bitcoin','7700k':'cpu','angular2':'framework','segwit2x':'bitcoin','hashflare':'bitcoin','940mx':'gpu',\n                'openai':'framework','hashflare':'bitcoin','1050ti':'gpu','nearbuy':'near buy','freebitco':'bitcoin','antminer':'bitcoin',\n                'filecoin':'bitcoin','whatapp':'whatsapp', 'empowr':'empower','1080ti':'gpu','crytocurrency':'cryptocurrency','8700k':'cpu',\n                'whatsaap':'whatsapp','g4560':'cpu','payymoney':'pay money','fuckboys':'fuck boys','intenship':'internship','zcash':'bitcoin',\n                'demonatisation':'demonetization','narcicist':'narcissist','mastuburation':'masturbation','trignometric':'trigonometric',\n                'cryptocurreny':'cryptocurrency','howdid':'how did','crytocurrencies':'cryptocurrencies','phycopath':'psychopath',\n                'bytecoin':'bitcoin','possesiveness':'possessiveness','scollege':'college','humanties':'humanities','altacoin':'bitcoin',\n                'demonitised':'demonetized','brasília':'brazilia','accolite':'accolyte','econimics':'economics','varrier':'warrier','quroa':'quora',\n                'statergy':'strategy','langague':'language', 'splatoon':'game','7600k':'cpu','gate2018':'gate 2018','in2018':'in 2018',\n                'narcassist':'narcissist','jiocoin':'bitcoin','hnlu':'hulu','7300hq':'cpu','weatern':'western','interledger':'blockchain',\n                'deplation':'deflation', 'cryptocurrencies':'cryptocurrency', 'bitcoin':'blockchain cryptocurrency',\n                \"ain't\": \"is not\",\"aren't\": \"are not\",\"can't\": \"cannot\",\"'cause\": \"because\",\"could've\": \"could have\",\"couldn't\": \"could not\", \n                \"didn't\": \"did not\",\"doesn't\": \"does not\",\"don't\": \"do not\",\"hadn't\": \"had not\",\"hasn't\": \"has not\",\"haven't\": \"have not\",\"he'd\": \"he would\",\n                \"he'll\": \"he will\",\"he's\": \"he is\",\"how'd\": \"how did\",\"how'd'y\": \"how do you\",\"how'll\": \"how will\",\"how's\": \"how is\",\"I'd\": \"I would\", \n                \"I'd've\": \"I would have\",\"I'll\": \"I will\",\"I'll've\": \"I will have\",\"I'm\": \"I am\",\"I've\": \"I have\", \"i'd\": \"i would\", \"i'd've\": \"i would have\", \n                \"i'll\": \"i will\",\"i'll've\": \"i will have\",\"i'm\": \"i am\",\"i've\": \"i have\",\"isn't\": \"is not\",\"it'd\": \"it would\", \"it'd've\": \"it would have\", \n                \"it'll\": \"it will\",\"it'll've\": \"it will have\",\"it's\": \"it is\", \"let's\": \"let us\",\"ma'am\": \"madam\",\"mayn't\": \"may not\",\"might've\": \"might have\",\n                \"mightn't\": \"might not\",\"mightn't've\": \"might not have\",\"must've\": \"must have\",\"mustn't\": \"must not\",\"mustn't've\": \"must not have\", \n                \"needn't\": \"need not\",\"needn't've\": \"need not have\",\"o'clock\": \"of the clock\",\"oughtn't\": \"ought not\",\"oughtn't've\": \"ought not have\", \n                \"shan't\": \"shall not\",\"sha'n't\": \"shall not\", \"shan't've\": \"shall not have\",\"she'd\": \"she would\", \"she'd've\": \"she would have\", \n                \"she'll\": \"she will\", \"she'll've\": \"she will have\",\"she's\": \"she is\", \"should've\": \"should have\",\"shouldn't\": \"should not\", \"shouldn't've\": \"should not have\", \n                \"so've\": \"so have\",\"so's\": \"so as\", \"this's\": \"this is\",\"that'd\": \"that would\", \"that'd've\": \"that would have\", \n                \"that's\": \"that is\", \"there'd\": \"there would\",\"there'd've\": \"there would have\", \"there's\": \"there is\", \n                \"here's\": \"here is\",\"they'd\": \"they would\", \"they'd've\": \"they would have\",\"they'll\": \"they will\", \"they'll've\": \"they will have\", \n                \"they're\": \"they are\", \"they've\": \"they have\",\"to've\": \"to have\", \"wasn't\": \"was not\",\"we'd\": \"we would\", \"we'd've\": \"we would have\", \n                \"we'll\": \"we will\", \"we'll've\": \"we will have\",\"we're\": \"we are\", \"we've\": \"we have\",\"weren't\": \"were not\", \"what'll\": \"what will\", \n                \"what'll've\": \"what will have\", \"what're\": \"what are\",\"what's\": \"what is\", \"what've\": \"what have\", \"when's\": \"when is\", \n                \"when've\": \"when have\", \"where'd\": \"where did\", \"where's\": \"where is\",\"where've\": \"where have\", \"who'll\": \"who will\", \"who'll've\": \"who will have\", \n                \"who's\": \"who is\", \"who've\": \"who have\", \"why's\": \"why is\", \"why've\": \"why have\",\"will've\": \"will have\", \"won't\": \"will not\", \"won't've\": \"will not have\", \n                \"would've\": \"would have\", \"wouldn't\": \"would not\", \"wouldn't've\": \"would not have\",\"y'all\": \"you all\", \"y'all'd\": \"you all would\",\"y'all'd've\": \"you all would have\",\n                \"y'all're\": \"you all are\",\"y'all've\": \"you all have\",\"you'd\": \"you would\",\"you'd've\": \"you would have\", \"you'll\": \"you will\", \"you'll've\": \"you will have\", \n                \"you're\": \"you are\", \"you've\": \"you have\"\n                }\n","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:42:09.612799Z","iopub.execute_input":"2022-01-08T22:42:09.613128Z","iopub.status.idle":"2022-01-08T22:42:09.683912Z","shell.execute_reply.started":"2022-01-08T22:42:09.613086Z","shell.execute_reply":"2022-01-08T22:42:09.682861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clear_punct(text):\n    for p in punctuation_list:\n        if p in text:\n            text = text.replace(p, \" \")\n    return text\n\nprint(clear_punct(\"Ha♌hEha.!fsd\"))\n\ndef replace_word(text):\n    for i in misspell_dict:\n        if i in text:\n            text = text.replace(i, misspell_dict[i])\n    return text \n\nprint(replace_word(\"you're my favourite narcicist\"))\n\ndef clear_num(text):\n    text = re.sub('[0-9]{5,}', '#####', text)\n    text = re.sub('[0-9]{4}', '####', text)\n    text = re.sub('[0-9]{3}', '###', text)\n    text = re.sub('[0-9]{2}', '##', text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:42:12.247302Z","iopub.execute_input":"2022-01-08T22:42:12.247644Z","iopub.status.idle":"2022-01-08T22:42:12.256507Z","shell.execute_reply.started":"2022-01-08T22:42:12.247613Z","shell.execute_reply":"2022-01-08T22:42:12.255667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[\"question_text\"] = train_data[\"question_text\"].apply(lambda x: x.lower())\ntrain_data[\"question_text\"] = train_data[\"question_text\"].apply(lambda x: clear_num(x))\ntrain_data[\"question_text\"] = train_data[\"question_text\"].apply(lambda x: replace_word(x))\ntrain_data[\"question_text\"] = train_data[\"question_text\"].apply(lambda x: clear_punct(x))\n\ntest_data[\"question_text\"] = test_data[\"question_text\"].apply(lambda x: x.lower())\ntest_data[\"question_text\"] = test_data[\"question_text\"].apply(lambda x: clear_num(x))\ntest_data[\"question_text\"] = test_data[\"question_text\"].apply(lambda x: replace_word(x))\ntest_data[\"question_text\"] = test_data[\"question_text\"].apply(lambda x: clear_punct(x))\n","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:42:15.20345Z","iopub.execute_input":"2022-01-08T22:42:15.204147Z","iopub.status.idle":"2022-01-08T22:44:44.985779Z","shell.execute_reply.started":"2022-01-08T22:42:15.204101Z","shell.execute_reply":"2022-01-08T22:44:44.98458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_x, val_x, train_y, val_y = train_test_split(train_data['question_text'].values, train_data['target'].values, random_state = 0)\n\ntest_x = test_data['question_text'].values\n","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:44:44.988007Z","iopub.execute_input":"2022-01-08T22:44:44.988349Z","iopub.status.idle":"2022-01-08T22:44:45.618879Z","shell.execute_reply.started":"2022-01-08T22:44:44.988286Z","shell.execute_reply":"2022-01-08T22:44:45.61785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data processing\n- Tokenizes the texts into words\n- Creates a vocabulary using the top 20,000 tokens\n- Converts the tokens into sequence vectors\n- Pads the sequences to a fixed sequence length","metadata":{}},{"cell_type":"code","source":"from tensorflow.python.keras.preprocessing import sequence\nfrom tensorflow.python.keras.preprocessing import text\n\n# Vectorization parameters\n# Limit on the number of features. We use the top 20K features.\n\n# Limit on the length of text sequences. Sequences longer than this\n# will be truncated.\nMAX_LENGTH = 100\n\n# Create vocabulary with training texts.\ntokenizer = text.Tokenizer(num_words=20000)\ntokenizer.fit_on_texts(list(train_x))\n\n# Vectorize training and validation texts.\ntrain_x = tokenizer.texts_to_sequences(train_x)\nval_x = tokenizer.texts_to_sequences(val_x)\ntest_x = tokenizer.texts_to_sequences(test_x)\n\n\n# Fix sequence length to max value. Sequences shorter than the length are\n# padded in the beginning and sequences longer are truncated\n# at the beginning.\ntrain_x = sequence.pad_sequences(train_x, maxlen=MAX_LENGTH)\nval_x = sequence.pad_sequences(val_x, maxlen=MAX_LENGTH)\ntest_x = sequence.pad_sequences(test_x, maxlen=MAX_LENGTH)\n","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:45:39.889882Z","iopub.execute_input":"2022-01-08T22:45:39.890777Z","iopub.status.idle":"2022-01-08T22:47:46.592254Z","shell.execute_reply.started":"2022-01-08T22:45:39.890733Z","shell.execute_reply":"2022-01-08T22:47:46.591445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from zipfile import ZipFile\n\nzip_file = ZipFile(emb_path, 'r')\nzip_file.extractall()\nzip_file.namelist()\n","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:47:46.593826Z","iopub.execute_input":"2022-01-08T22:47:46.594062Z","iopub.status.idle":"2022-01-08T22:51:54.432975Z","shell.execute_reply.started":"2022-01-08T22:47:46.594033Z","shell.execute_reply":"2022-01-08T22:51:54.432051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"glove_path = './glove.840B.300d/glove.840B.300d.txt'\nwiki_path = './wiki-news-300d-1M/wiki-news-300d-1M.vec'\nparagram_path =  './paragram_300_sl999/paragram_300_sl999.txt'\n","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:53:46.109514Z","iopub.execute_input":"2022-01-08T22:53:46.110209Z","iopub.status.idle":"2022-01-08T22:53:46.114728Z","shell.execute_reply.started":"2022-01-08T22:53:46.110163Z","shell.execute_reply":"2022-01-08T22:53:46.113892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_features = 20000\nemb_dim = 300","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:53:48.960781Z","iopub.execute_input":"2022-01-08T22:53:48.961368Z","iopub.status.idle":"2022-01-08T22:53:48.966265Z","shell.execute_reply.started":"2022-01-08T22:53:48.961331Z","shell.execute_reply":"2022-01-08T22:53:48.965642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/gmhost/gru-capsule\n# paragram uses utf8, different from others -> this function looks stupid\ndef load_emb(word_index, emb_path, is_parag=False):\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    if is_parag:\n        emb_index = dict(get_coefs(*o.split(\" \")) for o in open(emb_path, encoding=\"utf8\", errors='ignore') if len(o)>100 and o.split(\" \")[0] in word_index)\n    else:\n        emb_index = dict(get_coefs(*o.split(\" \")) for o in open(emb_path) if len(o)>100 and o.split(\" \")[0] in word_index )\n\n    all_embs = np.stack(emb_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    emb_matrix = np.random.normal(emb_mean, emb_std, (num_features, embed_size))\n    for word, i in word_index.items():\n        if i >= num_features: continue\n        emb_vector = emb_index.get(word)\n        if emb_vector is not None: emb_matrix[i] = emb_vector\n\n    return emb_matrix\n\n","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:53:52.863655Z","iopub.execute_input":"2022-01-08T22:53:52.864166Z","iopub.status.idle":"2022-01-08T22:53:52.875077Z","shell.execute_reply.started":"2022-01-08T22:53:52.864117Z","shell.execute_reply":"2022-01-08T22:53:52.874371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"glove_emb = load_emb(tokenizer.word_index, glove_path)\nwiki_emb = load_emb(tokenizer.word_index, wiki_path)\nparag_emb = load_emb(tokenizer.word_index, paragram_path, True)\n\nemb_matrix = np.mean([glove_emb, wiki_emb, parag_emb], axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:53:55.043495Z","iopub.execute_input":"2022-01-08T22:53:55.044275Z","iopub.status.idle":"2022-01-08T22:56:55.830259Z","shell.execute_reply.started":"2022-01-08T22:53:55.044229Z","shell.execute_reply":"2022-01-08T22:56:55.829458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model\nUse pretrained embeddings","metadata":{}},{"cell_type":"code","source":"from tensorflow.python.keras import models\nfrom tensorflow.keras.callbacks import EarlyStopping\n\nfrom tensorflow.python.keras.layers import Dense\nfrom tensorflow.python.keras.layers import Input\nfrom tensorflow.python.keras.layers import Dropout\nfrom tensorflow.python.keras.layers import Embedding\nfrom tensorflow.python.keras.layers import Bidirectional\nfrom tensorflow.python.keras.layers import LSTM\nfrom tensorflow.python.keras.layers import Conv1D\nfrom tensorflow.python.keras.layers import GlobalMaxPool1D\nfrom tensorflow.python.keras.layers import GRU\n\ninputs = Input(shape=(MAX_LENGTH))\nx = Embedding(input_dim=num_features, output_dim=emb_dim,weights=[emb_matrix], trainable=False)(inputs)\nx = Bidirectional(GRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\noutputs = Dense(1, activation=\"sigmoid\")(x)\n\nmodel = models.Model(inputs=inputs, outputs=outputs)\nmodel.compile(loss=\"binary_crossentropy\", optimizer=\"adam\", metrics=[\"accuracy\"])\n# callbacks = [EarlyStopping(monitor='val_loss', patience=2)]\n\nmodel.summary()\n","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:56:55.831786Z","iopub.execute_input":"2022-01-08T22:56:55.832019Z","iopub.status.idle":"2022-01-08T22:56:56.18137Z","shell.execute_reply.started":"2022-01-08T22:56:55.831989Z","shell.execute_reply":"2022-01-08T22:56:56.179611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_x,train_y,epochs=5,validation_data=(val_x,val_y), verbose=1,batch_size=128)\n\n# Print results.\nhistory = history.history\nprint('Validation accuracy: {acc}, loss: {loss}'.format(\n        acc=history['val_acc'][-1], loss=history['val_loss'][-1]))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-01-08T22:57:01.957537Z","iopub.execute_input":"2022-01-08T22:57:01.957848Z","iopub.status.idle":"2022-01-09T06:45:29.839682Z","shell.execute_reply.started":"2022-01-08T22:57:01.957815Z","shell.execute_reply":"2022-01-09T06:45:29.837191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_x, batch_size = 1024)\n\ny_test = (pred[:,0] > 0.5).astype(np.int)\nsubmit = pd.DataFrame({\"qid\": test_data[\"qid\"], \"prediction\": y_test})\nsubmit.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-09T06:50:10.14394Z","iopub.execute_input":"2022-01-09T06:50:10.144375Z","iopub.status.idle":"2022-01-09T06:53:42.316476Z","shell.execute_reply.started":"2022-01-09T06:50:10.144336Z","shell.execute_reply":"2022-01-09T06:53:42.315534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}