{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"},{"sourceId":4798524,"sourceType":"datasetVersion","datasetId":2778076}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h2 style=\"text-align:center;font-size:200%;\">\n    <b>Toxic Text Detection using BERT&FineTuning</b>\n</h2>\n<h3  style=\"text-align:center;\">Keywords : \n    <span style=\"border-radius:7px;background-color:limegreen;color:white;padding:7px;\">NLP</span>\n    <span style=\"border-radius:7px;background-color:limegreen;color:white;padding:7px;\">EDA</span>\n    <span style=\"border-radius:7px;background-color:limegreen;color:white;padding:7px;\">Text Mining</span>\n    <span style=\"border-radius:7px;background-color:limegreen;color:white;padding:7px;\">PyTorch</span>\n    <span style=\"border-radius:7px;background-color:limegreen;color:white;padding:7px;\">BERT</span>\n    <span style=\"border-radius:7px;background-color:limegreen;color:white;padding:7px;\">HuggingFace</span>\n</h3>\n\n<hr>\n\n<a id='top'></a>\n<h2 style=\"font-size:150%;\"><span id='top'>Table of Contents</span></h2>\n<blockquote>\n    <ol>\n        <li><a href=\"#Overview\">Overview</a></li>\n        <li><a href=\"#Preparation\">Preparation</a></li>\n        <li><a href=\"#Data-Load\">Data Load</a></li>\n        <li><a href=\"#Pre-Processing\">Pre-Processing</a></li>\n        <li><a href=\"#EDA\">EDA</a></li>\n        <ul>\n            <li><a href=\"#Sincere/Insincere-Distribution\">Sincere/Insincere Distribution</a></li>\n            <li><a href=\"#Word-Length-Distribution\">Word Length Distribution</a></li>\n            <li><a href=\"#Sentence-Distribution\">Sentence Distribution</a></li>\n            <li><a href=\"#Punctuation-Distribution\">Punctuation Distribution</a></li>\n            <li><a href=\"#Word-Frequencies\">Word Frequencies</a></li>\n            <li><a href=\"#WordCloud\">WordCloud</a></li>\n        </ul>\n        <li><a href=\"#Modeling\">Modeling</a></li>\n        <ul>\n            <li><a href=\"#Model-Settings\">Model Settings</a></li>\n            <li><a href=\"#Data-Preparation\">Data Preparation</a></li>\n            <li><a href=\"#Training\">Training</a></li>\n            <li><a href=\"#Evaluation\">Evaluation</a></li>\n        </ul>\n        <li><a href=\"#Modeling2\">Modeling2</a></li>\n        <ul>\n            <li><a href=\"#Load-Dataset-from-Pandas\">Load Dataset from Pandas</a></li>\n            <li><a href=\"#Tokenize\">Tokenize</a></li>\n            <li><a href=\"#Batch-Setting\">Batch Setting</a></li>\n            <li><a href=\"#Model-Setting\">Model Setting</a></li>\n            <li><a href=\"#LoRA\">LoRA</a></li>\n            <li><a href=\"#Evaluation\">Evaluation</a></li>\n        </ul>\n        <li><a href=\"#Conclusion\">Conclusion</a></li>\n        <li><a href=\"#References\">References</a></li>\n        <li><a href=\"#Submission\">Submission</a></li>\n    </ol>\n</blockquote>","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"text-align: left; background-color: mediumseagreen; color: white; padding: 10px; line-height:1;border-radius:10px\">Overview</div>\n## About the Dataset\n- Questions in [Quora](https://www.quora.com/)\n- Questions is classified in sincere or not\n    - characteristics of <b>insincere</b> questions are:\n        - Has a non-neutral tone\n        - Is disparaging or inflammatory\n        - Isn't grounded in reality\n        - Uses sexual content (incest, bestiality, pedophilia) for shock value, and not to seek genuine answers\n        \n## What You Get from this Notebook\n- ETL technique\n- EDA & Visualization\n- Classification Modeling using Pytorch\n    - By building a model that can detect insincere questions, we can save time and effort in the operation of Quora and help maintain a safer speech space.","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"text-align: left; background-color: mediumseagreen; color: white; padding: 10px; line-height:1;border-radius:10px\">Preparation</div>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\npd.set_option(\"display.max_colwidth\", 80)\nimport os\nimport matplotlib.pyplot as plt\nimport holoviews as hv\nfrom holoviews import opts\nhv.extension('bokeh')\nimport string\n\nimport collections\nimport itertools\nimport nltk\nfrom nltk.corpus import stopwords\nnltk.download('stopwords')\nimport spacy\nfrom wordcloud import WordCloud\nfrom PIL import Image\nimport requests\nimport io\nfrom tqdm import tqdm\ntqdm.pandas()","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:15:56.467674Z","iopub.execute_input":"2024-05-12T10:15:56.468298Z","iopub.status.idle":"2024-05-12T10:16:07.279924Z","shell.execute_reply.started":"2024-05-12T10:15:56.468268Z","shell.execute_reply":"2024-05-12T10:16:07.279117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<button class=\"label alert-success\" style=\"border-radius:10px;padding:10px;font-size:18px\"><a href=\"#top\" style=\"color:green;\"><b>Table of Contents</b></a></button>","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"text-align: left; background-color: mediumseagreen; color: white; padding: 10px; line-height:1;border-radius:10px\">Data Load</div>","metadata":{}},{"cell_type":"code","source":"data_dir = \"/kaggle/input\"","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:16:07.281832Z","iopub.execute_input":"2024-05-12T10:16:07.282777Z","iopub.status.idle":"2024-05-12T10:16:07.286803Z","shell.execute_reply.started":"2024-05-12T10:16:07.282742Z","shell.execute_reply":"2024-05-12T10:16:07.285819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _, filenames in os.walk(f\"{data_dir}/quora-insincere-questions-classification\"):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:16:07.287802Z","iopub.execute_input":"2024-05-12T10:16:07.288085Z","iopub.status.idle":"2024-05-12T10:16:07.301075Z","shell.execute_reply.started":"2024-05-12T10:16:07.288063Z","shell.execute_reply":"2024-05-12T10:16:07.300119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(f\"{data_dir}/quora-insincere-questions-classification/train.csv\")\nprint('Train Set Shape = {}'.format(train.shape))\ntrain.head(3)","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_kg_hide-output":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:16:07.303387Z","iopub.execute_input":"2024-05-12T10:16:07.303741Z","iopub.status.idle":"2024-05-12T10:16:11.229164Z","shell.execute_reply.started":"2024-05-12T10:16:07.303711Z","shell.execute_reply":"2024-05-12T10:16:11.228297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(f\"{data_dir}/quora-insincere-questions-classification/test.csv\")\nprint('Test Set Shape = {}'.format(test.shape))\ntest.head(3)","metadata":{"_kg_hide-output":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:16:11.230224Z","iopub.execute_input":"2024-05-12T10:16:11.230511Z","iopub.status.idle":"2024-05-12T10:16:12.293644Z","shell.execute_reply.started":"2024-05-12T10:16:11.230467Z","shell.execute_reply":"2024-05-12T10:16:12.292554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(f\"{data_dir}/quora-insincere-questions-classification/sample_submission.csv\")\nsubmission.head(3)","metadata":{"_kg_hide-output":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:16:12.295211Z","iopub.execute_input":"2024-05-12T10:16:12.295500Z","iopub.status.idle":"2024-05-12T10:16:12.647300Z","shell.execute_reply.started":"2024-05-12T10:16:12.295460Z","shell.execute_reply":"2024-05-12T10:16:12.646456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<button class=\"label alert-success\" style=\"border-radius:10px;padding:10px;font-size:18px\"><a href=\"#top\" style=\"color:green;\"><b>Table of Contents</b></a></button>","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"text-align: left; background-color: mediumseagreen; color: white; padding: 10px; line-height:1;border-radius:10px\">Pre-Processing</div>","metadata":{}},{"cell_type":"markdown","source":"## Count Words","metadata":{}},{"cell_type":"code","source":"train[\"word_num\"] = train[\"question_text\"].apply(lambda x: len(x.split()))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-12T10:16:12.648368Z","iopub.execute_input":"2024-05-12T10:16:12.648643Z","iopub.status.idle":"2024-05-12T10:16:14.554770Z","shell.execute_reply.started":"2024-05-12T10:16:12.648620Z","shell.execute_reply":"2024-05-12T10:16:14.553969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Count Punctuations","metadata":{}},{"cell_type":"code","source":"string.punctuation","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:16:14.555781Z","iopub.execute_input":"2024-05-12T10:16:14.556065Z","iopub.status.idle":"2024-05-12T10:16:14.561636Z","shell.execute_reply.started":"2024-05-12T10:16:14.556040Z","shell.execute_reply":"2024-05-12T10:16:14.560774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"punc_list\"] = train[\"question_text\"].apply(lambda x: [i for i in x if i in string.punctuation] )","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-12T10:16:14.562953Z","iopub.execute_input":"2024-05-12T10:16:14.563298Z","iopub.status.idle":"2024-05-12T10:16:24.758748Z","shell.execute_reply.started":"2024-05-12T10:16:14.563236Z","shell.execute_reply":"2024-05-12T10:16:24.757908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"punc_cnt\"] = train[\"punc_list\"].apply(lambda x: len(x))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-12T10:16:24.762101Z","iopub.execute_input":"2024-05-12T10:16:24.762385Z","iopub.status.idle":"2024-05-12T10:16:25.569325Z","shell.execute_reply.started":"2024-05-12T10:16:24.762355Z","shell.execute_reply":"2024-05-12T10:16:25.568538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Text Processing","metadata":{}},{"cell_type":"code","source":"# stopwords_en = stopwords.words(\"english\")\n# nlp = spacy.load('en_core_web_sm') #en_core_web_lg\ndef en_preprocess(text):\n    lowered = text.lower()\n#     tokenized = []\n#     for i in nlp(lowered):\n#         word = i.lemma_\n#         pos = i.pos_\n#         if i.is_alpha and pos not in ['ADV','PRON','CCONJ','PUNCT','PART','DET','ADP','SPACE'] and i.text not in stopwords_en:\n#             tokenized.append(word)\n#     preprocessed = \" \".join(tokenized)\n    return lowered","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:16:25.570470Z","iopub.execute_input":"2024-05-12T10:16:25.570754Z","iopub.status.idle":"2024-05-12T10:16:25.575670Z","shell.execute_reply.started":"2024-05-12T10:16:25.570730Z","shell.execute_reply":"2024-05-12T10:16:25.574670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"text_preproc\"] = train[\"question_text\"].progress_apply(lambda x: en_preprocess(x))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-12T10:16:25.577528Z","iopub.execute_input":"2024-05-12T10:16:25.577783Z","iopub.status.idle":"2024-05-12T10:16:27.825278Z","shell.execute_reply.started":"2024-05-12T10:16:25.577762Z","shell.execute_reply":"2024-05-12T10:16:27.824533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get Mean Word Length","metadata":{}},{"cell_type":"code","source":"train[\"mean_word_len\"] = train[\"text_preproc\"].progress_apply(lambda x: np.mean([len(i) for i in x.split()]) )","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-12T10:16:27.826622Z","iopub.execute_input":"2024-05-12T10:16:27.826903Z","iopub.status.idle":"2024-05-12T10:16:52.048810Z","shell.execute_reply.started":"2024-05-12T10:16:27.826880Z","shell.execute_reply":"2024-05-12T10:16:52.047859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Get Sentence Length","metadata":{}},{"cell_type":"code","source":"train[\"sentence_len\"] = train[\"question_text\"].progress_apply(lambda x: len(x))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-12T10:16:52.050005Z","iopub.execute_input":"2024-05-12T10:16:52.050334Z","iopub.status.idle":"2024-05-12T10:16:54.355512Z","shell.execute_reply.started":"2024-05-12T10:16:52.050303Z","shell.execute_reply":"2024-05-12T10:16:54.354608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train.to_csv(f\"{data_dir}/quora_train.csv\",index=False)\ntrain = pd.read_csv(f\"/kaggle/input/quara-train/quora_train.csv\") # Load Pre-Processed dataframe\ntrain.head()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:16:54.356841Z","iopub.execute_input":"2024-05-12T10:16:54.357692Z","iopub.status.idle":"2024-05-12T10:17:01.269352Z","shell.execute_reply.started":"2024-05-12T10:16:54.357654Z","shell.execute_reply":"2024-05-12T10:17:01.268503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<button class=\"label alert-success\" style=\"border-radius:10px;padding:10px;font-size:18px\"><a href=\"#top\" style=\"color:green;\"><b>Table of Contents</b></a></button>","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"text-align: left; background-color: mediumseagreen; color: white; padding: 10px; line-height:1;border-radius:10px\">EDA</div>","metadata":{}},{"cell_type":"markdown","source":"## Sincere/Insincere Distribution\nThe majority of the questions are sincere.","metadata":{}},{"cell_type":"code","source":"hv.Bars( train.target.value_counts(normalize=True).to_frame().rename(index={0:\"sincere\",1:\"insincere\"})*100 )\\\n.opts(xlabel=\"Target\",ylabel=\"Count\",yformatter=\"%d%%\", \\\n      width=400,height=300,title=\"Sincere / Insincere Counts\",tools=['hover'],fontsize={'xticks': 10})","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:01.270624Z","iopub.execute_input":"2024-05-12T10:17:01.270958Z","iopub.status.idle":"2024-05-12T10:17:01.454988Z","shell.execute_reply.started":"2024-05-12T10:17:01.270931Z","shell.execute_reply":"2024-05-12T10:17:01.454087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Question Examples","metadata":{}},{"cell_type":"markdown","source":"<b>Sincere Questions</b>","metadata":{}},{"cell_type":"code","source":"for i in train[train.target==0].question_text.values[0:5]:\n    print(f\"- {i}\")","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:01.456290Z","iopub.execute_input":"2024-05-12T10:17:01.456714Z","iopub.status.idle":"2024-05-12T10:17:01.596614Z","shell.execute_reply.started":"2024-05-12T10:17:01.456682Z","shell.execute_reply":"2024-05-12T10:17:01.595759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<b>Insincere Questions</b>","metadata":{}},{"cell_type":"code","source":"for i in train[train.target==1].question_text.values[0:5]:\n    print(f\"- {i}\")","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:01.597969Z","iopub.execute_input":"2024-05-12T10:17:01.598703Z","iopub.status.idle":"2024-05-12T10:17:01.635830Z","shell.execute_reply.started":"2024-05-12T10:17:01.598669Z","shell.execute_reply":"2024-05-12T10:17:01.634894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Word Length Distribution\nSincere questions tend to use longer words.","metadata":{}},{"cell_type":"code","source":"_sin_mean = np.round(train[train.target==0].mean_word_len.mean(),2)\n_insin_mean = np.round(train[train.target==1].mean_word_len.mean(),2)\nd = train[[\"target\",\"mean_word_len\"]].copy()\nd.target.replace({0:\"sincere\", 1:\"insincere\"},inplace=True)\ng = hv.BoxWhisker(d, kdims=[\"target\"], vdims='mean_word_len')\\\n    * hv.Text(\"sincere\", 35, f\"Avg: {_sin_mean}\") * hv.Text(\"insincere\", 35, f\"Avg: {_insin_mean}\")\ng.opts(opts.BoxWhisker(title=\"Sincere / Insincere Words Length\", xlabel=\"\",ylabel=\"Length\",\\\n    width=400, height=300,tools=['hover'],fontsize={'xticks': 10}))","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:01.637006Z","iopub.execute_input":"2024-05-12T10:17:01.637286Z","iopub.status.idle":"2024-05-12T10:17:10.369058Z","shell.execute_reply.started":"2024-05-12T10:17:01.637263Z","shell.execute_reply":"2024-05-12T10:17:10.368279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sentence Distribution","metadata":{}},{"cell_type":"code","source":"_sin_mean = np.round(train[train.target==0].sentence_len.mean(),2)\n_insin_mean = np.round(train[train.target==1].sentence_len.mean(),2)\nd = train[[\"target\",\"sentence_len\"]].copy()\nd.target.replace({0:\"sincere\", 1:\"insincere\"},inplace=True)\ng = hv.BoxWhisker(d, kdims=[\"target\"], vdims='sentence_len')\\\n    * hv.Text(\"sincere\", 600, f\"Avg: {_sin_mean}\") * hv.Text(\"insincere\", 600, f\"Avg: {_insin_mean}\")\ng.opts(opts.BoxWhisker(title=\"Sincere / Insincere Sentence Length\", xlabel=\"\",ylabel=\"Length\",\\\n    width=400, height=300,tools=['hover'],fontsize={'xticks': 10}))","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:10.370433Z","iopub.execute_input":"2024-05-12T10:17:10.371026Z","iopub.status.idle":"2024-05-12T10:17:19.557301Z","shell.execute_reply.started":"2024-05-12T10:17:10.370995Z","shell.execute_reply":"2024-05-12T10:17:19.556147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Punctuation Distribution","metadata":{}},{"cell_type":"code","source":"_sin_mean = np.round(train[train.target==0].punc_cnt.mean(),2)\n_insin_mean = np.round(train[train.target==1].punc_cnt.mean(),2)\nd = train[[\"target\",\"punc_cnt\"]].copy()\nd.target.replace({0:\"sincere\", 1:\"insincere\"},inplace=True)\ng = hv.BoxWhisker(d, kdims=[\"target\"], vdims='punc_cnt')\\\n    * hv.Text(\"sincere\", 250, f\"Avg: {_sin_mean}\") * hv.Text(\"insincere\", 250, f\"Avg: {_insin_mean}\")\ng.opts(opts.BoxWhisker(title=\"Sincere / Insincere Punctuations Count\", xlabel=\"\",ylabel=\"Count\",\\\n    width=400, height=300,tools=['hover'],fontsize={'xticks': 10}))","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:19.558382Z","iopub.execute_input":"2024-05-12T10:17:19.559000Z","iopub.status.idle":"2024-05-12T10:17:28.998105Z","shell.execute_reply.started":"2024-05-12T10:17:19.558967Z","shell.execute_reply":"2024-05-12T10:17:28.996638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<button class=\"label alert-success\" style=\"border-radius:10px;padding:10px;font-size:18px\"><a href=\"#top\" style=\"color:green;\"><b>Table of Contents</b></a></button>","metadata":{}},{"cell_type":"markdown","source":"## Word Frequencies","metadata":{}},{"cell_type":"code","source":"_sincere = train[train.target==0].dropna()\n_insincere = train[train.target==1].dropna()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:28.999680Z","iopub.execute_input":"2024-05-12T10:17:28.999985Z","iopub.status.idle":"2024-05-12T10:17:29.771178Z","shell.execute_reply.started":"2024-05-12T10:17:28.999960Z","shell.execute_reply":"2024-05-12T10:17:29.770398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ngram_func(ngram, text_series):\n    # string_filterd =  text_series.sum().split()\n    string_filterd =  \" \".join(text_series).split()\n    dic = nltk.FreqDist(nltk.ngrams(string_filterd, ngram)).most_common(30)\n    ngram_df = pd.DataFrame(dic, columns=['ngram','count'])\n    ngram_df.index = [' '.join(i) for i in ngram_df.ngram]\n    ngram_df.drop('ngram',axis=1, inplace=True)\n    return ngram_df","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:29.772278Z","iopub.execute_input":"2024-05-12T10:17:29.772570Z","iopub.status.idle":"2024-05-12T10:17:29.778422Z","shell.execute_reply.started":"2024-05-12T10:17:29.772546Z","shell.execute_reply":"2024-05-12T10:17:29.777537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Unigram\n<b>The insincere questions are more likely to be political, racial, and words about specific persons.</b>","metadata":{}},{"cell_type":"code","source":"d1 = ngram_func(1, _sincere[\"text_preproc\"].values)\nd2 = ngram_func(1, _insincere[\"text_preproc\"].values)\ng1 = hv.Bars(d1[0:20][::-1]).opts(title=\"Sincere Unigram - Top20\")\ng2 = hv.Bars(d2[0:20][::-1]).opts(title=\"Insincere Unigram - Top20\", color=\"red\")\n(g1 + g2).opts(opts.Bars(xlabel=\"Unigram\", ylabel=\"Count\", width=400, height=400,tools=['hover'],\\\n    show_grid=True ,invert_axes=True,fontsize={'yticks': 9})).opts(shared_axes=False)","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:29.779630Z","iopub.execute_input":"2024-05-12T10:17:29.779891Z","iopub.status.idle":"2024-05-12T10:17:41.733435Z","shell.execute_reply.started":"2024-05-12T10:17:29.779869Z","shell.execute_reply":"2024-05-12T10:17:41.732536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Bigram","metadata":{}},{"cell_type":"code","source":"d1 = ngram_func(2, _sincere[\"text_preproc\"].values)\nd2 = ngram_func(2, _insincere[\"text_preproc\"].values)\ng1 = hv.Bars(d1[0:20][::-1]).opts(title=\"Sincere Bigram - Top20\")\ng2 = hv.Bars(d2[0:20][::-1]).opts(title=\"Insincere Bigram - Top20\", color=\"red\")\n(g1 + g2).opts(opts.Bars(xlabel=\"Bigram\", ylabel=\"Count\", width=400, height=400,tools=['hover'],\\\n    show_grid=True ,invert_axes=True,fontsize={'yticks': 9})).opts(shared_axes=False)","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:41.734877Z","iopub.execute_input":"2024-05-12T10:17:41.735306Z","iopub.status.idle":"2024-05-12T10:17:57.658452Z","shell.execute_reply.started":"2024-05-12T10:17:41.735274Z","shell.execute_reply":"2024-05-12T10:17:57.657529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Punctuations\n<b>It can be seen that insincere questions contain more exclamation marks (\"!\").</b>","metadata":{}},{"cell_type":"code","source":"_sin = collections.Counter(list(itertools.chain.from_iterable(_sincere.punc_list.values))).most_common(20)\n_insin = collections.Counter(list(itertools.chain.from_iterable(_insincere.punc_list.values))).most_common(20)\n_sin_punc = pd.DataFrame(_sin).rename(columns={0:\"punc\", 1:\"cnt\"}).set_index(\"punc\")[::-1]\n_sin_punc = ( _sin_punc/_sin_punc.sum() ) * 100\n_insin_punc = pd.DataFrame(_insin).rename(columns={0:\"punc\", 1:\"cnt\"}).set_index(\"punc\")[::-1]\n_insin_punc = ( _insin_punc/_insin_punc.sum() ) * 100\ng1 = hv.Bars( _sin_punc ).opts(title=\"Sincere Punctuations - Top20\")\ng2 = hv.Bars( _insin_punc ).opts(title=\"Insincere Punctuations - Top20\", color=\"red\")\n(g1 + g2).opts(opts.Bars(xlabel=\"Punctuations\", ylabel=\"Count\", width=400, height=400,tools=['hover'],\\\n    show_grid=True ,invert_axes=True,fontsize={'yticks': 10})).opts(shared_axes=False)","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:57.659730Z","iopub.execute_input":"2024-05-12T10:17:57.660031Z","iopub.status.idle":"2024-05-12T10:17:58.851428Z","shell.execute_reply.started":"2024-05-12T10:17:57.660005Z","shell.execute_reply":"2024-05-12T10:17:58.850475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<button class=\"label alert-success\" style=\"border-radius:10px;padding:10px;font-size:18px\"><a href=\"#top\" style=\"color:green;\"><b>Table of Contents</b></a></button>","metadata":{}},{"cell_type":"markdown","source":"## WordCloud","metadata":{}},{"cell_type":"code","source":"_sin_text = \" \".join( _sincere[\"text_preproc\"].values )\n_insin_text = \" \".join( _insincere[\"text_preproc\"].values )\n_image_url = \"https://cdn-icons-png.flaticon.com/512/174/174865.png\"","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:58.852810Z","iopub.execute_input":"2024-05-12T10:17:58.853163Z","iopub.status.idle":"2024-05-12T10:17:59.045518Z","shell.execute_reply.started":"2024-05-12T10:17:58.853131Z","shell.execute_reply":"2024-05-12T10:17:59.044678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def wordcloud_mask(img_url, text, _title, _cmap):\n    _img = io.BytesIO(requests.get(img_url).content)\n    _mask_img = np.array(Image.open( _img ))\n    wordcloud = WordCloud(background_color='white',width=800, height=600, \\\n        min_font_size=1, max_words=None, collocations=False, min_word_length=1,  \\\n        mask=_mask_img, contour_width =1, contour_color=\"#f5f5f5\", colormap=_cmap)\n    wordcloud.generate(text)\n    plt.figure(figsize=(10, 10))\n    plt.title(_title, fontsize=20)\n    plt.imshow(wordcloud, interpolation=\"bilinear\")\n    plt.show()","metadata":{"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:59.052765Z","iopub.execute_input":"2024-05-12T10:17:59.053089Z","iopub.status.idle":"2024-05-12T10:17:59.060531Z","shell.execute_reply.started":"2024-05-12T10:17:59.053057Z","shell.execute_reply":"2024-05-12T10:17:59.059772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wordcloud_mask(_image_url, _sin_text, \"Sincere Questions\", \"Blues\")","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:17:59.061634Z","iopub.execute_input":"2024-05-12T10:17:59.061989Z","iopub.status.idle":"2024-05-12T10:18:18.990159Z","shell.execute_reply.started":"2024-05-12T10:17:59.061960Z","shell.execute_reply":"2024-05-12T10:18:18.989146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wordcloud_mask(_image_url, _insin_text, \"Insincere Questions\", \"Reds\")","metadata":{"_kg_hide-input":true,"tags":[],"execution":{"iopub.status.busy":"2024-05-12T10:18:18.991164Z","iopub.execute_input":"2024-05-12T10:18:18.991425Z","iopub.status.idle":"2024-05-12T10:18:27.047888Z","shell.execute_reply.started":"2024-05-12T10:18:18.991403Z","shell.execute_reply":"2024-05-12T10:18:27.046930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<button class=\"label alert-success\" style=\"border-radius:10px;padding:10px;font-size:18px\"><a href=\"#top\" style=\"color:green;\"><b>Table of Contents</b></a></button>","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"text-align: left; background-color: mediumseagreen; color: white; padding: 10px; line-height:1;border-radius:10px\">Modeling</div>\n\n<b>This section is inspired by the source code of <a href=\"https://www.kaggle.com/code/lemonwaffle/quora-pytorch-torchtext\">this notebook</a>. Thanks!</b>","metadata":{}},{"cell_type":"markdown","source":"Load the relevant pytorch libraries","metadata":{}},{"cell_type":"code","source":"! pip install transformers[ja,sentencepiece,torch] datasets  matplotlib peft wandb","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-12T10:18:27.049274Z","iopub.execute_input":"2024-05-12T10:18:27.049728Z","iopub.status.idle":"2024-05-12T10:19:09.715512Z","shell.execute_reply.started":"2024-05-12T10:18:27.049691Z","shell.execute_reply":"2024-05-12T10:19:09.714377Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoTokenizer\nfrom transformers import AutoModel\nimport torch\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score, accuracy_score\nfrom tqdm import tqdm\nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:19:09.716977Z","iopub.execute_input":"2024-05-12T10:19:09.717254Z","iopub.status.idle":"2024-05-12T10:19:11.370320Z","shell.execute_reply.started":"2024-05-12T10:19:09.717230Z","shell.execute_reply":"2024-05-12T10:19:11.369537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preparation","metadata":{}},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(\"bert-base-uncased\")","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-12T10:19:11.371544Z","iopub.execute_input":"2024-05-12T10:19:11.373112Z","iopub.status.idle":"2024-05-12T10:19:13.153747Z","shell.execute_reply.started":"2024-05-12T10:19:11.373074Z","shell.execute_reply":"2024-05-12T10:19:13.152888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"define dataset class","metadata":{}},{"cell_type":"code","source":"class Data(Dataset):\n    def __init__(self, data, tokenizer, max_length, test=False):\n        self.data = data\n        self.tokenizer = tokenizer\n        self.max_length = max_length\n        self.test = test\n    \n    def __len__(self):\n        return len(self.data)\n    \n    def __getitem__(self, idx):\n        text = self.data[\"question_text\"].values[idx]\n        encoded = tokenizer(\n            text,\n            padding = \"max_length\",\n            max_length = self.max_length,\n            truncation = True\n        )\n        qids = self.data[\"qid\"].values[idx]\n        input_ids = torch.tensor(encoded[\"input_ids\"], dtype = torch.int32)\n        attention_mask = torch.tensor(encoded[\"attention_mask\"], dtype = torch.int32)\n        token_type_ids = torch.tensor(encoded[\"token_type_ids\"], dtype = torch.int32)\n        \n        if self.test:\n            return qids, input_ids, attention_mask, token_type_ids\n        \n        label = self.data[\"target\"].values[idx]\n        label = torch.tensor(label, dtype = torch.int32)\n        return qids, input_ids, attention_mask, token_type_ids, label","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:19:13.154931Z","iopub.execute_input":"2024-05-12T10:19:13.155196Z","iopub.status.idle":"2024-05-12T10:19:13.164426Z","shell.execute_reply.started":"2024-05-12T10:19:13.155173Z","shell.execute_reply":"2024-05-12T10:19:13.163394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- make dataset  \n- this time, training dataset is too large (and time-consuming) to try out some training process, so I used only top-**10000** dataset for train/valid.","metadata":{}},{"cell_type":"code","source":"MAX_LEN = 256\ntrain_data, valid_data = train_test_split(train[:10000], train_size=0.7)\ntrain_ds = Data(train_data, tokenizer, MAX_LEN)\nvalid_ds = Data(valid_data, tokenizer, MAX_LEN)\ntest_ds = Data(test, tokenizer, MAX_LEN, test=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:19:13.165571Z","iopub.execute_input":"2024-05-12T10:19:13.165843Z","iopub.status.idle":"2024-05-12T10:19:13.180705Z","shell.execute_reply.started":"2024-05-12T10:19:13.165819Z","shell.execute_reply":"2024-05-12T10:19:13.179856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"make dataloader","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 50\ntrain_dl = DataLoader(train_ds, batch_size = BATCH_SIZE, shuffle = True, drop_last = True, num_workers=os.cpu_count(), pin_memory=True)\nvalid_dl = DataLoader(valid_ds, batch_size = BATCH_SIZE, shuffle = False, drop_last = False, num_workers=os.cpu_count(), pin_memory=True)\ntest_dl = DataLoader(test_ds, batch_size = BATCH_SIZE, shuffle = False, drop_last = False, num_workers=os.cpu_count(), pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:19:13.181829Z","iopub.execute_input":"2024-05-12T10:19:13.182098Z","iopub.status.idle":"2024-05-12T10:19:13.189414Z","shell.execute_reply.started":"2024-05-12T10:19:13.182075Z","shell.execute_reply":"2024-05-12T10:19:13.188467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"example output","metadata":{}},{"cell_type":"code","source":"next(iter(train_dl)) # qid, input_ids, attention_mask , token_type_ids , label","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-12T10:19:13.190643Z","iopub.execute_input":"2024-05-12T10:19:13.191276Z","iopub.status.idle":"2024-05-12T10:19:13.669961Z","shell.execute_reply.started":"2024-05-12T10:19:13.191248Z","shell.execute_reply":"2024-05-12T10:19:13.668858Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<button class=\"label alert-success\" style=\"border-radius:10px;padding:10px;font-size:18px\"><a href=\"#top\" style=\"color:green;\"><b>Table of Contents</b></a></button>","metadata":{}},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"markdown","source":"### Model","metadata":{}},{"cell_type":"code","source":"class Model(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.bert = AutoModel.from_pretrained(\"bert-base-uncased\")\n        self.dropout = nn.Dropout(p=0.2)\n        \n        self.fc1 = nn.Linear(in_features = 768, out_features = 512)\n        self.fc2 = nn.Linear(in_features = 512, out_features = 2)\n    \n    def forward(self, input_ids, attention_mask, token_type_ids):\n        x = self.bert(\n            input_ids = input_ids, attention_mask = attention_mask, token_type_ids = token_type_ids\n        )\n        x = self.fc1(self.dropout(x.pooler_output))\n        x = self.fc2(x).squeeze(-1)\n        return x\n\nmodel = Model().cuda()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-12T10:19:13.671539Z","iopub.execute_input":"2024-05-12T10:19:13.671926Z","iopub.status.idle":"2024-05-12T10:19:17.243862Z","shell.execute_reply.started":"2024-05-12T10:19:13.671888Z","shell.execute_reply":"2024-05-12T10:19:17.242954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<button class=\"label alert-success\" style=\"border-radius:10px;padding:10px;font-size:18px\"><a href=\"#top\" style=\"color:green;\"><b>Table of Contents</b></a></button>","metadata":{}},{"cell_type":"markdown","source":"### Run","metadata":{}},{"cell_type":"markdown","source":"define early-stopping function","metadata":{}},{"cell_type":"code","source":"class EarlyStopping:\n    def __init__(self, tolerance=5, min_delta=0):\n\n        self.tolerance = tolerance\n        self.min_delta = min_delta\n        self.counter = 0\n        self.stop = False\n\n    def __call__(self, train_loss, validation_loss):\n        if (validation_loss - train_loss) > self.min_delta:\n            self.counter +=1\n            if self.counter >= self.tolerance:  \n                self.stop = True","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:19:17.245030Z","iopub.execute_input":"2024-05-12T10:19:17.245585Z","iopub.status.idle":"2024-05-12T10:19:17.251940Z","shell.execute_reply.started":"2024-05-12T10:19:17.245557Z","shell.execute_reply":"2024-05-12T10:19:17.250701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = torch.optim.Adam(model.parameters(), lr = 1e-5)\ncriterion =  nn.CrossEntropyLoss()\nearly_stopping = EarlyStopping(tolerance=3, min_delta=0.1)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:19:17.253194Z","iopub.execute_input":"2024-05-12T10:19:17.253502Z","iopub.status.idle":"2024-05-12T10:19:17.280031Z","shell.execute_reply.started":"2024-05-12T10:19:17.253461Z","shell.execute_reply":"2024-05-12T10:19:17.279119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_loss = np.inf\nEPOCHS = 2\n\nfor epoch in range(EPOCHS):\n    print(f\"EPOCH[{epoch}]\")\n    \n    model.train()\n    train_loss = 0\n    for batch in tqdm(train_dl):\n        optimizer.zero_grad()\n        input_ids, attention_mask, token_type_ids, label = batch[1].cuda(), batch[2].cuda(), batch[3].cuda(), batch[4].long().cuda()\n        preds = model(input_ids, attention_mask, token_type_ids)\n        loss = criterion(preds, label)\n        loss.backward()\n        optimizer.step()\n        train_loss += loss.item()\n    train_loss /= len(train_dl)\n\n    model.eval()\n    valid_loss = 0\n    with torch.no_grad():\n        for batch in tqdm(valid_dl):\n            input_ids, attention_mask, token_type_ids, label = batch[1].cuda(), batch[2].cuda(), batch[3].cuda(), batch[4].long().cuda()\n            preds = model(input_ids, attention_mask, token_type_ids)\n            loss = criterion(preds, label)\n            valid_loss += loss.item()\n        valid_loss /= len(valid_dl)\n    \n    print(f\"TRAIN LOSS: {train_loss}\")\n    print(f\"VALID LOSS: {valid_loss}\")\n    \n    early_stopping(train_loss, valid_loss)\n    if early_stopping.stop:\n        print(f\"Early Stopped at epoch:{epoch}\")\n        break\n    \n    if valid_loss < best_loss:\n        best_loss = valid_loss\n\nprint(f\"### BEST LOSS: {best_loss} ###\")","metadata":{"tags":[],"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-05-12T10:19:17.281245Z","iopub.execute_input":"2024-05-12T10:19:17.281555Z","iopub.status.idle":"2024-05-12T10:25:27.811891Z","shell.execute_reply.started":"2024-05-12T10:19:17.281530Z","shell.execute_reply":"2024-05-12T10:25:27.810527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluation","metadata":{}},{"cell_type":"code","source":"pred1 = []\nmodel.eval()\nwith torch.no_grad():\n    for batch in tqdm(valid_dl):\n        input_ids, attention_mask, token_type_ids = batch[1].cuda(), batch[2].cuda(), batch[3].cuda()\n        logits = model(input_ids, attention_mask, token_type_ids).cpu().numpy()\n        pred1.extend(logits.argmax(axis = 1))","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:25:27.813778Z","iopub.execute_input":"2024-05-12T10:25:27.814174Z","iopub.status.idle":"2024-05-12T10:25:51.806799Z","shell.execute_reply.started":"2024-05-12T10:25:27.814134Z","shell.execute_reply":"2024-05-12T10:25:51.804930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"truth = valid_data[\"target\"].values\nprint(classification_report(truth, pred1))","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:25:51.808488Z","iopub.execute_input":"2024-05-12T10:25:51.808828Z","iopub.status.idle":"2024-05-12T10:25:51.832613Z","shell.execute_reply.started":"2024-05-12T10:25:51.808797Z","shell.execute_reply":"2024-05-12T10:25:51.831540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<button class=\"label alert-success\" style=\"border-radius:10px;padding:10px;font-size:18px\"><a href=\"#top\" style=\"color:green;\"><b>Table of Contents</b></a></button>","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"text-align: left; background-color: mediumseagreen; color: white; padding: 10px; line-height:1;border-radius:10px\">Modeling2</div>","metadata":{}},{"cell_type":"code","source":"from transformers import pipeline\nfrom transformers import AutoTokenizer\nfrom transformers.trainer_utils import set_seed\nset_seed(42)\nfrom transformers import BatchEncoding\nfrom transformers import DataCollatorWithPadding\nfrom transformers import TrainingArguments\nfrom transformers import Trainer\nfrom transformers import AutoModelForSequenceClassification\nfrom transformers import EarlyStoppingCallback\nimport numpy as np\nimport peft\nfrom datasets import DatasetDict\nfrom datasets import Dataset\nos.environ[\"WANDB_DISABLED\"] = \"true\"","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:34:43.716193Z","iopub.execute_input":"2024-05-12T10:34:43.716626Z","iopub.status.idle":"2024-05-12T10:34:43.723650Z","shell.execute_reply.started":"2024-05-12T10:34:43.716594Z","shell.execute_reply":"2024-05-12T10:34:43.722524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, valid_data = train_test_split(train[:30000], train_size=0.7, stratify=train[:30000]['target'])","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:01.925268Z","iopub.execute_input":"2024-05-12T10:26:01.926275Z","iopub.status.idle":"2024-05-12T10:26:01.954172Z","shell.execute_reply.started":"2024-05-12T10:26:01.926239Z","shell.execute_reply":"2024-05-12T10:26:01.953169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:01.955470Z","iopub.execute_input":"2024-05-12T10:26:01.955839Z","iopub.status.idle":"2024-05-12T10:26:01.970084Z","shell.execute_reply.started":"2024-05-12T10:26:01.955802Z","shell.execute_reply":"2024-05-12T10:26:01.969143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Dataset from Pandas","metadata":{}},{"cell_type":"code","source":"train_dataset = Dataset.from_pandas(train_data[[\"question_text\",\"target\"]])\nvalid_dataset = Dataset.from_pandas(valid_data[[\"question_text\",\"target\"]])","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:01.971535Z","iopub.execute_input":"2024-05-12T10:26:01.971981Z","iopub.status.idle":"2024-05-12T10:26:02.023113Z","shell.execute_reply.started":"2024-05-12T10:26:01.971948Z","shell.execute_reply":"2024-05-12T10:26:02.022337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tokenize","metadata":{}},{"cell_type":"code","source":"model_name = \"distilbert/distilbert-base-uncased-finetuned-sst-2-english\"  #\"bert-base-uncased\"\ntokenizer = AutoTokenizer.from_pretrained(model_name)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:02.024197Z","iopub.execute_input":"2024-05-12T10:26:02.024514Z","iopub.status.idle":"2024-05-12T10:26:03.312356Z","shell.execute_reply.started":"2024-05-12T10:26:02.024467Z","shell.execute_reply":"2024-05-12T10:26:03.311522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def text_tokenizer( inputs: dict[str, str | int] ) -> BatchEncoding:\n    encoded_inputs = tokenizer(inputs[\"question_text\"], max_length=512)\n    encoded_inputs[\"labels\"] = inputs[\"target\"]\n    return encoded_inputs","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:03.313708Z","iopub.execute_input":"2024-05-12T10:26:03.314360Z","iopub.status.idle":"2024-05-12T10:26:03.319859Z","shell.execute_reply.started":"2024-05-12T10:26:03.314315Z","shell.execute_reply":"2024-05-12T10:26:03.318553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_train_dataset = train_dataset.map(\n    text_tokenizer,\n    remove_columns=train_dataset.column_names,\n)\nencoded_valid_dataset = valid_dataset.map(\n    text_tokenizer,\n    remove_columns=valid_dataset.column_names,\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:03.320955Z","iopub.execute_input":"2024-05-12T10:26:03.321229Z","iopub.status.idle":"2024-05-12T10:26:10.134799Z","shell.execute_reply.started":"2024-05-12T10:26:03.321205Z","shell.execute_reply":"2024-05-12T10:26:10.133873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_train_dataset[0]","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:10.136281Z","iopub.execute_input":"2024-05-12T10:26:10.136762Z","iopub.status.idle":"2024-05-12T10:26:10.144848Z","shell.execute_reply.started":"2024-05-12T10:26:10.136725Z","shell.execute_reply":"2024-05-12T10:26:10.143941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer.convert_ids_to_tokens(encoded_train_dataset[0]['input_ids'])","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:10.146160Z","iopub.execute_input":"2024-05-12T10:26:10.146495Z","iopub.status.idle":"2024-05-12T10:26:10.156781Z","shell.execute_reply.started":"2024-05-12T10:26:10.146469Z","shell.execute_reply":"2024-05-12T10:26:10.156012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Batch Setting","metadata":{}},{"cell_type":"code","source":"data_collator = DataCollatorWithPadding(tokenizer=tokenizer)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:10.157980Z","iopub.execute_input":"2024-05-12T10:26:10.158273Z","iopub.status.idle":"2024-05-12T10:26:10.166593Z","shell.execute_reply.started":"2024-05-12T10:26:10.158249Z","shell.execute_reply":"2024-05-12T10:26:10.165803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Setting","metadata":{}},{"cell_type":"code","source":"label2id = { \"0\":0 ,\"1\":1  }\nid2label = { 0:\"0\", 1:\"1\" }\nbase_model = AutoModelForSequenceClassification.from_pretrained(\n    model_name,\n    num_labels=2,\n    label2id=label2id,\n    id2label=id2label\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:10.167899Z","iopub.execute_input":"2024-05-12T10:26:10.168252Z","iopub.status.idle":"2024-05-12T10:26:11.821222Z","shell.execute_reply.started":"2024-05-12T10:26:10.168219Z","shell.execute_reply":"2024-05-12T10:26:11.820394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LoRA","metadata":{}},{"cell_type":"code","source":"peft_config = peft.LoraConfig(\n    task_type=peft.TaskType.SEQ_CLS, \n    r=8, \n    lora_alpha=32, \n    lora_dropout=0.1, \n    inference_mode=False,\n    target_modules = \"all-linear\"\n)\nlora_model = peft.get_peft_model(base_model, peft_config)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:11.822417Z","iopub.execute_input":"2024-05-12T10:26:11.822737Z","iopub.status.idle":"2024-05-12T10:26:11.914110Z","shell.execute_reply.started":"2024-05-12T10:26:11.822711Z","shell.execute_reply":"2024-05-12T10:26:11.913201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lora_model.print_trainable_parameters()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:26:11.915194Z","iopub.execute_input":"2024-05-12T10:26:11.915472Z","iopub.status.idle":"2024-05-12T10:26:11.922050Z","shell.execute_reply.started":"2024-05-12T10:26:11.915447Z","shell.execute_reply":"2024-05-12T10:26:11.921096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_args = TrainingArguments(\n    output_dir=\"./outputs\",\n    per_device_train_batch_size=64,\n    per_device_eval_batch_size=64,\n    learning_rate=5e-5,\n    lr_scheduler_type=\"linear\",\n    warmup_ratio=0.1,\n    num_train_epochs=20,\n    save_strategy=\"epoch\",\n    logging_strategy=\"epoch\",\n    evaluation_strategy=\"epoch\",\n    load_best_model_at_end=True,\n    metric_for_best_model=\"f1\",\n    fp16=True,\n    report_to='none',\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:35:40.854160Z","iopub.execute_input":"2024-05-12T10:35:40.854620Z","iopub.status.idle":"2024-05-12T10:35:40.880589Z","shell.execute_reply.started":"2024-05-12T10:35:40.854586Z","shell.execute_reply":"2024-05-12T10:35:40.879563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compute_accuracy(\n    eval_pred: tuple[np.ndarray, np.ndarray]\n) -> dict[str, float]:\n    predictions, labels = eval_pred\n    predictions = np.argmax(predictions, axis=1)\n    f1 = f1_score(labels, predictions, average='macro')\n\n    return {\"accuracy\": (predictions == labels).mean(), \"f1\":f1 }","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:35:46.223317Z","iopub.execute_input":"2024-05-12T10:35:46.224285Z","iopub.status.idle":"2024-05-12T10:35:46.230050Z","shell.execute_reply.started":"2024-05-12T10:35:46.224249Z","shell.execute_reply":"2024-05-12T10:35:46.228959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer = Trainer(\n    model=lora_model,\n    train_dataset=encoded_train_dataset,\n    eval_dataset=encoded_valid_dataset,\n    data_collator=data_collator,\n    args=training_args,\n    compute_metrics=compute_accuracy,\n    callbacks=[EarlyStoppingCallback(early_stopping_patience=3)],\n)\ntrainer.train()","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:39:59.381085Z","iopub.execute_input":"2024-05-12T10:39:59.382036Z","iopub.status.idle":"2024-05-12T10:44:44.753187Z","shell.execute_reply.started":"2024-05-12T10:39:59.382000Z","shell.execute_reply":"2024-05-12T10:44:44.752186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluation","metadata":{}},{"cell_type":"code","source":"trainer.evaluate(encoded_valid_dataset)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:45:03.953558Z","iopub.execute_input":"2024-05-12T10:45:03.954560Z","iopub.status.idle":"2024-05-12T10:45:12.917051Z","shell.execute_reply.started":"2024-05-12T10:45:03.954517Z","shell.execute_reply":"2024-05-12T10:45:12.916118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_cls = pipeline(\"text-classification\", model=lora_model, tokenizer=tokenizer, return_all_scores=False)","metadata":{"_kg_hide-output":true,"_kg_hide-input":false,"execution":{"iopub.status.busy":"2024-05-12T10:45:30.510994Z","iopub.execute_input":"2024-05-12T10:45:30.511395Z","iopub.status.idle":"2024-05-12T10:45:30.674479Z","shell.execute_reply.started":"2024-05-12T10:45:30.511350Z","shell.execute_reply":"2024-05-12T10:45:30.673574Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prediction(text):\n    predict_result = text_cls(text)[0]\n    highest_score = predict_result\n    return label2id[highest_score['label']]","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:45:42.109246Z","iopub.execute_input":"2024-05-12T10:45:42.110013Z","iopub.status.idle":"2024-05-12T10:45:42.117172Z","shell.execute_reply.started":"2024-05-12T10:45:42.109954Z","shell.execute_reply":"2024-05-12T10:45:42.115958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_data[\"pred\"] = valid_data['question_text'].progress_apply(lambda x: prediction(x) )","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:45:46.198949Z","iopub.execute_input":"2024-05-12T10:45:46.199965Z","iopub.status.idle":"2024-05-12T10:55:45.986334Z","shell.execute_reply.started":"2024-05-12T10:45:46.199930Z","shell.execute_reply":"2024-05-12T10:55:45.985292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"truth = valid_data[\"target\"].values\npred2 = valid_data[\"pred\"].values\nprint(classification_report(truth, pred2))","metadata":{"execution":{"iopub.status.busy":"2024-05-12T11:07:50.376532Z","iopub.execute_input":"2024-05-12T11:07:50.377366Z","iopub.status.idle":"2024-05-12T11:07:50.411729Z","shell.execute_reply.started":"2024-05-12T11:07:50.377324Z","shell.execute_reply":"2024-05-12T11:07:50.410758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <div style=\"text-align: left; background-color: mediumseagreen; color: white; padding: 10px; line-height:1;border-radius:10px\">Conclusion</div>","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-success\" role=\"alert\" style=\"border-radius:10px\">\n<ul>\n    <li>From the EDA, we found certain trends in insincere questions.</li>\n    <ul>\n        <li>words length</li>\n        <li>sentence length</li>\n        <li>punctuation usage</li>\n        <li>words usage</li>\n    </ul>\n    <li>The classification model by pytorch enables the detection of insincere questions with a certain degree of accuracy</li>\n    <li>Fine-tuned lightweight LLM models were also achieved with a good degree of accuracy.</li>\n</ul>\n</div>","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"text-align: left; background-color: mediumseagreen; color: white; padding: 10px; line-height:1;border-radius:10px\">References</div>\n- refenreced notebooks\n    - EDA\n        - https://www.kaggle.com/code/aryanml007/quora-insincere-eda-lstm-gru-embeddings\n    \n    - Modeling\n        - https://www.kaggle.com/code/arnikaer/pytorch\n        - https://www.kaggle.com/code/christofhenkel/how-to-preprocessing-when-using-embeddings\n        - https://www.kaggle.com/code/lemonwaffle/quora-pytorch-torchtext\n- Pytorch\n    - https://pytorch.org/text/stable/index.html\n    - https://pytorch.org/docs/stable/index.html\n    - early stopping technique\n        - https://stackoverflow.com/questions/71998978/early-stopping-in-pytorch\n    - dataloader acceleration technique\n        - https://qiita.com/sugulu_Ogawa_ISID/items/62f5f7adee083d96a587#11-num_workers\n- Transformer based modeling\n    - https://htomblog.com/python-bert\n    - https://www.kaggle.com/code/bearmontblanc/nbme-pytorch-bert\n    - bert output \n        - https://huggingface.co/docs/transformers/main_classes/output\n    - how to explore in dataloader\n        - https://output-zakki.com/dataloader_iter_and_next/\n    - how to fine-tune\n        - https://www.ai-shift.co.jp/techblog/2145\n- Case study\n    - https://techblog.yahoo.co.jp/entry/2021122030233811/\n- HuggingFace\n    - Pipieline\n        - https://huggingface.co/docs/transformers/ja/main_classes/pipelines#transformers.TextClassificationPipeline\n    - Load Dataset from Pandas\n        - https://dev.classmethod.jp/articles/huggingface-usage-dataset/","metadata":{}},{"cell_type":"markdown","source":"# <div style=\"text-align: left; background-color: mediumseagreen; color: white; padding: 10px; line-height:1;border-radius:10px\">Submission</div>","metadata":{}},{"cell_type":"markdown","source":"submission process takes time, so I commented out submission code.","metadata":{}},{"cell_type":"code","source":"# sub = []\n# ids = []\n# model.eval()\n# with torch.no_grad():\n#     for batch in tqdm(test_dl):\n#         qid, input_ids, attention_mask, token_type_ids = batch[0], batch[1].cuda(), batch[2].cuda(), batch[3].cuda()\n#         logits = model(input_ids, attention_mask, token_type_ids).cpu().numpy()\n#         pred_labels = logits.argmax(axis = 1)\n#         sub.extend(pred_labels)\n#         ids.extend(qid)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:34:29.775575Z","iopub.status.idle":"2024-05-12T10:34:29.776036Z","shell.execute_reply.started":"2024-05-12T10:34:29.775798Z","shell.execute_reply":"2024-05-12T10:34:29.775818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission = pd.DataFrame({'qid': ids, 'prediction': sub})\n# submission.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-12T10:34:29.777179Z","iopub.status.idle":"2024-05-12T10:34:29.777654Z","shell.execute_reply.started":"2024-05-12T10:34:29.777400Z","shell.execute_reply":"2024-05-12T10:34:29.777420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<button class=\"label alert-success\" style=\"border-radius:10px;padding:10px;font-size:18px\"><a href=\"#top\" style=\"color:green;\"><b>Table of Contents</b></a></button>","metadata":{}}]}