{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<p style = \"font-size:40px; \nfont-family: Helvetica; \nfont-weight : bold; \nbackground-color: #036EB7; \ncolor : #FFFFFF; \ntext-align: left; \npadding: 0px 15px; \nborder-radius:3px\">\n\tJigsaw All-in-One Dataset\n</p>","metadata":{"_uuid":"ab318b03-3704-4768-a919-60ced1962bda","_cell_guid":"9b9b54ac-947c-4458-9cea-227f6356cc70","jupyter":{"outputs_hidden":false}}},{"cell_type":"markdown","source":"My goal is creating all-in-one dataset for jigsaw competition.  \nFirst, I concatenated **'toxic comment classification challenge'**s dataset, **'jigsaw unintended bias in toxicity classification'**s dataset.   \nIf I find more new effective dataset, I'll update this notebook.  ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os.path as osp","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:26:58.021092Z","iopub.execute_input":"2021-12-02T03:26:58.02145Z","iopub.status.idle":"2021-12-02T03:26:58.046115Z","shell.execute_reply.started":"2021-12-02T03:26:58.021363Z","shell.execute_reply":"2021-12-02T03:26:58.04546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"INPUT_PATH = '/kaggle/input/'","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:26:58.047735Z","iopub.execute_input":"2021-12-02T03:26:58.047947Z","iopub.status.idle":"2021-12-02T03:26:58.051731Z","shell.execute_reply.started":"2021-12-02T03:26:58.047922Z","shell.execute_reply":"2021-12-02T03:26:58.05073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style = \"font-size:25px; \nfont-family: Helvetica; \nfont-weight : normal; \nbackground-color: #036EB7; \ncolor : #FFFFFF; \ntext-align: left; \npadding: 0px 15px; \nborder-radius:3px\">\n    [Toxic Comment Classification Challenge] Data\n</p>","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:45:47.784748Z","iopub.execute_input":"2021-12-01T12:45:47.785008Z","iopub.status.idle":"2021-12-01T12:45:47.791885Z","shell.execute_reply.started":"2021-12-01T12:45:47.78498Z","shell.execute_reply":"2021-12-01T12:45:47.790661Z"}}},{"cell_type":"markdown","source":"### [Reference : https://www.kaggle.com/c/jigsaw-toxic-comment-classification-challenge/data](https://www.kaggle.com/c/jigsaw-toxic-comment-classification-challenge/data)\n\nYou are provided with a large number of Wikipedia comments which have been labeled by human raters for toxic behavior. The types of toxicity are:\n\n* `toxic`\n* `severe_toxic`\n* `obscene`\n* `threat`\n* `insult`\n* `identity_hate`\n\nYou must create a model which predicts a probability of each type of toxicity for each comment.\n\nFile descriptions  \n* **train.csv** - the training set, contains comments with their binary labels\n* **test.csv** - the test set, you must predict the toxicity probabilities for these comments. To deter hand labeling, the test set contains some comments which are not included in scoring.\n* **sample_submission.csv** - a sample submission file in the correct format\n* **test_labels.csv** - labels for the test data; value of -1 indicates it was not used for scoring; (Note: file added after competition close!)\n","metadata":{}},{"cell_type":"code","source":"!unzip -n /kaggle/input/jigsaw-toxic-comment-classification-challenge/train.csv.zip -d ./ \n!unzip -n /kaggle/input/jigsaw-toxic-comment-classification-challenge/test.csv.zip -d ./\n!unzip -n /kaggle/input/jigsaw-toxic-comment-classification-challenge/test_labels.csv.zip -d ./","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:26:58.109222Z","iopub.execute_input":"2021-12-02T03:26:58.109715Z","iopub.status.idle":"2021-12-02T03:27:02.305522Z","shell.execute_reply.started":"2021-12-02T03:26:58.109681Z","shell.execute_reply":"2021-12-02T03:27:02.304695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('train.csv')\ntest = pd.read_csv('test.csv')\ntest_labels = pd.read_csv('test_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:02.307422Z","iopub.execute_input":"2021-12-02T03:27:02.309261Z","iopub.status.idle":"2021-12-02T03:27:04.537898Z","shell.execute_reply.started":"2021-12-02T03:27:02.309214Z","shell.execute_reply":"2021-12-02T03:27:04.537262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:04.538862Z","iopub.execute_input":"2021-12-02T03:27:04.539237Z","iopub.status.idle":"2021-12-02T03:27:04.561339Z","shell.execute_reply.started":"2021-12-02T03:27:04.539195Z","shell.execute_reply":"2021-12-02T03:27:04.560418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:04.562413Z","iopub.execute_input":"2021-12-02T03:27:04.562647Z","iopub.status.idle":"2021-12-02T03:27:04.572821Z","shell.execute_reply.started":"2021-12-02T03:27:04.56262Z","shell.execute_reply":"2021-12-02T03:27:04.571837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_labels.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:04.575821Z","iopub.execute_input":"2021-12-02T03:27:04.57613Z","iopub.status.idle":"2021-12-02T03:27:04.58998Z","shell.execute_reply.started":"2021-12-02T03:27:04.57609Z","shell.execute_reply":"2021-12-02T03:27:04.589172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_labels.describe()","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:04.591337Z","iopub.execute_input":"2021-12-02T03:27:04.592342Z","iopub.status.idle":"2021-12-02T03:27:04.655147Z","shell.execute_reply.started":"2021-12-02T03:27:04.592297Z","shell.execute_reply":"2021-12-02T03:27:04.654359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are some -1 labels. Delete that.","metadata":{}},{"cell_type":"code","source":"test = test.merge(test_labels, on=\"id\")\ntest = test[test.toxic != -1]\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:04.656781Z","iopub.execute_input":"2021-12-02T03:27:04.657382Z","iopub.status.idle":"2021-12-02T03:27:04.814022Z","shell.execute_reply.started":"2021-12-02T03:27:04.657338Z","shell.execute_reply":"2021-12-02T03:27:04.813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.describe()","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:04.815198Z","iopub.execute_input":"2021-12-02T03:27:04.815417Z","iopub.status.idle":"2021-12-02T03:27:04.848009Z","shell.execute_reply.started":"2021-12-02T03:27:04.815387Z","shell.execute_reply":"2021-12-02T03:27:04.847214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cool!","metadata":{}},{"cell_type":"code","source":"toxic_comment = pd.concat([train, test])","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:04.849321Z","iopub.execute_input":"2021-12-02T03:27:04.849535Z","iopub.status.idle":"2021-12-02T03:27:04.866862Z","shell.execute_reply.started":"2021-12-02T03:27:04.84951Z","shell.execute_reply":"2021-12-02T03:27:04.866128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_comment.describe()","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:04.867904Z","iopub.execute_input":"2021-12-02T03:27:04.868258Z","iopub.status.idle":"2021-12-02T03:27:04.919133Z","shell.execute_reply.started":"2021-12-02T03:27:04.868217Z","shell.execute_reply":"2021-12-02T03:27:04.918561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style = \"font-size:25px; \nfont-family: Helvetica; \nfont-weight : normal; \nbackground-color: #036EB7; \ncolor : #FFFFFF; \ntext-align: left; \npadding: 0px 15px; \nborder-radius:3px\">\n   [Jigsaw Unintended Bias in Toxicity Classification] Data\n</p>","metadata":{"execution":{"iopub.status.busy":"2021-12-01T12:38:56.581549Z","iopub.execute_input":"2021-12-01T12:38:56.581907Z","iopub.status.idle":"2021-12-01T12:38:56.616134Z","shell.execute_reply.started":"2021-12-01T12:38:56.581862Z","shell.execute_reply":"2021-12-01T12:38:56.615182Z"}}},{"cell_type":"markdown","source":"### [Reference : https://www.kaggle.com/c/jigsaw-unintended-bias-in-toxicity-classification/data](https://www.kaggle.com/c/jigsaw-unintended-bias-in-toxicity-classification/data)\n\nAt the end of 2017 the Civil Comments platform shut down and chose make their ~2m public comments from their platform available in a lasting open archive so that researchers could understand and improve civility in online conversations for years to come. Jigsaw sponsored this effort and extended annotation of this data by human raters for various toxic conversational attributes.\n\nIn the data supplied for this competition, the text of the individual comment is found in the comment_text column. Each comment in Train has a toxicity label (target), and models should predict the target toxicity for the Test data. This attribute (and all others) are fractional values which represent the fraction of human raters who believed the attribute applied to the given comment. For evaluation, test set examples with target >= 0.5 will be considered to be in the positive class (toxic).\n\nThe data also has several additional toxicity subtype attributes. Models do not need to predict these attributes for the competition, they are included as an additional avenue for research. Subtype attributes are:\n\n* `severe_toxicity`\n* `obscene`\n* `threat`\n* `insult`\n* `identity_attack`\n* `sexual_explicit`","metadata":{}},{"cell_type":"code","source":"!ls /kaggle/input/jigsaw-unintended-bias-in-toxicity-classification/","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:04.920124Z","iopub.execute_input":"2021-12-02T03:27:04.920545Z","iopub.status.idle":"2021-12-02T03:27:05.721555Z","shell.execute_reply.started":"2021-12-02T03:27:04.920517Z","shell.execute_reply":"2021-12-02T03:27:05.720433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We need `all_data.csv`.","metadata":{"execution":{"iopub.status.busy":"2021-12-01T13:24:41.289491Z","iopub.execute_input":"2021-12-01T13:24:41.290455Z","iopub.status.idle":"2021-12-01T13:24:41.296425Z","shell.execute_reply.started":"2021-12-01T13:24:41.29041Z","shell.execute_reply":"2021-12-01T13:24:41.295286Z"}}},{"cell_type":"code","source":"!cp /kaggle/input/jigsaw-unintended-bias-in-toxicity-classification/all_data.csv ./unintended.csv","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:05.723243Z","iopub.execute_input":"2021-12-02T03:27:05.723518Z","iopub.status.idle":"2021-12-02T03:27:15.550512Z","shell.execute_reply.started":"2021-12-02T03:27:05.723487Z","shell.execute_reply":"2021-12-02T03:27:15.54952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unintended = pd.read_csv('unintended.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:15.551719Z","iopub.execute_input":"2021-12-02T03:27:15.551931Z","iopub.status.idle":"2021-12-02T03:27:33.210749Z","shell.execute_reply.started":"2021-12-02T03:27:15.551906Z","shell.execute_reply":"2021-12-02T03:27:33.209653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unintended.columns","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.214663Z","iopub.execute_input":"2021-12-02T03:27:33.215143Z","iopub.status.idle":"2021-12-02T03:27:33.22171Z","shell.execute_reply.started":"2021-12-02T03:27:33.215103Z","shell.execute_reply":"2021-12-02T03:27:33.22076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unintended.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.222989Z","iopub.execute_input":"2021-12-02T03:27:33.223299Z","iopub.status.idle":"2021-12-02T03:27:33.251568Z","shell.execute_reply.started":"2021-12-02T03:27:33.223261Z","shell.execute_reply":"2021-12-02T03:27:33.250739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unintended.columns","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.252996Z","iopub.execute_input":"2021-12-02T03:27:33.253433Z","iopub.status.idle":"2021-12-02T03:27:33.260065Z","shell.execute_reply.started":"2021-12-02T03:27:33.253392Z","shell.execute_reply":"2021-12-02T03:27:33.259527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_columns = [\n    \"id\", \"comment_text\", \"toxicity\", \"severe_toxicity\", \n    \"obscene\", \"identity_attack\", \"insult\", \"threat\"\n]\nunintended = unintended[target_columns]","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.26141Z","iopub.execute_input":"2021-12-02T03:27:33.261623Z","iopub.status.idle":"2021-12-02T03:27:33.358453Z","shell.execute_reply.started":"2021-12-02T03:27:33.261596Z","shell.execute_reply":"2021-12-02T03:27:33.357525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unintended.describe()","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.35994Z","iopub.execute_input":"2021-12-02T03:27:33.360337Z","iopub.status.idle":"2021-12-02T03:27:33.763046Z","shell.execute_reply.started":"2021-12-02T03:27:33.360304Z","shell.execute_reply":"2021-12-02T03:27:33.762441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style = \"font-size:25px; \nfont-family: Helvetica; \nfont-weight : normal; \nbackground-color: #036EB7; \ncolor : #FFFFFF; \ntext-align: left; \npadding: 0px 15px; \nborder-radius:3px\">\n    Merge\n</p>","metadata":{}},{"cell_type":"code","source":"base_columns = [\"toxic\", \"severe_toxic\", \"obscene\", \"threat\", \"insult\", \"identity_hate\"]\nintended_columns = [\"toxicity\", \"severe_toxicity\", \"obscene\", \"threat\", \"insult\", \"identity_attack\"]","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.763948Z","iopub.execute_input":"2021-12-02T03:27:33.764626Z","iopub.status.idle":"2021-12-02T03:27:33.768907Z","shell.execute_reply.started":"2021-12-02T03:27:33.764594Z","shell.execute_reply":"2021-12-02T03:27:33.767854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unintended = unintended.rename(columns = {x: y for x, y in zip(intended_columns, base_columns)})","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.769837Z","iopub.execute_input":"2021-12-02T03:27:33.770042Z","iopub.status.idle":"2021-12-02T03:27:33.879387Z","shell.execute_reply.started":"2021-12-02T03:27:33.770011Z","shell.execute_reply":"2021-12-02T03:27:33.878524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('length of jigsaw-toxic-comment-classification-challenge \\'s dataset :', len(toxic_comment))","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.88076Z","iopub.execute_input":"2021-12-02T03:27:33.881602Z","iopub.status.idle":"2021-12-02T03:27:33.887719Z","shell.execute_reply.started":"2021-12-02T03:27:33.881565Z","shell.execute_reply":"2021-12-02T03:27:33.886783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('length of jigsaw-unintended-bias-in-toxicity-classification \\'s dataset :', len(unintended))","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.889487Z","iopub.execute_input":"2021-12-02T03:27:33.889802Z","iopub.status.idle":"2021-12-02T03:27:33.902632Z","shell.execute_reply.started":"2021-12-02T03:27:33.88976Z","shell.execute_reply":"2021-12-02T03:27:33.901658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_comment['dataset'] = 'toxic_comment'\nunintended['dataset'] = 'unintended'","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.903929Z","iopub.execute_input":"2021-12-02T03:27:33.904157Z","iopub.status.idle":"2021-12-02T03:27:33.924736Z","shell.execute_reply.started":"2021-12-02T03:27:33.904133Z","shell.execute_reply":"2021-12-02T03:27:33.924066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = pd.concat([toxic_comment, unintended])\nfinal.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:33.925593Z","iopub.execute_input":"2021-12-02T03:27:33.925812Z","iopub.status.idle":"2021-12-02T03:27:34.241301Z","shell.execute_reply.started":"2021-12-02T03:27:33.925785Z","shell.execute_reply":"2021-12-02T03:27:34.24035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's a final dataset. (toxic_comment + unintended)","metadata":{}},{"cell_type":"markdown","source":"<p style = \"font-size:25px; \nfont-family: Helvetica; \nfont-weight : normal; \nbackground-color: #036EB7; \ncolor : #FFFFFF; \ntext-align: left; \npadding: 0px 15px; \nborder-radius:3px\">\n    Toxic Data Preprocessing\n</p>","metadata":{}},{"cell_type":"markdown","source":"### Reference : [https://www.kaggle.com/fizzbuzz/toxic-data-preprocessing](https://www.kaggle.com/fizzbuzz/toxic-data-preprocessing)","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport copy\nimport re\nfrom keras.preprocessing.text import text_to_word_sequence\nfrom nltk import WordNetLemmatizer\n\n\nclass BaseTokenizer(object):\n    def process_text(self, text):\n        raise NotImplemented\n\n    def process(self, texts):\n        for text in texts:\n            yield self.process_text(text)\n\n\nRE_PATTERNS = {\n    ' american ':\n        [\n            'amerikan'\n        ],\n    ' adolf ':\n        [\n            'adolf'\n        ],\n    ' hitler ':\n        [\n            'hitler'\n        ],\n    ' fuck':\n        [\n            '(f)(u|[^a-z0-9 ])(c|[^a-z0-9 ])(k|[^a-z0-9 ])([^ ])*',\n            '(f)([^a-z]*)(u)([^a-z]*)(c)([^a-z]*)(k)',\n            ' f[!@#\\$%\\^\\&\\*]*u[!@#\\$%\\^&\\*]*k', 'f u u c',\n            '(f)(c|[^a-z ])(u|[^a-z ])(k)', r'f\\*',\n            'feck ', ' fux ', 'f\\*\\*', \n            'f\\-ing', 'f\\.u\\.', 'f###', ' fu ', 'f@ck', 'f u c k', 'f uck', 'f ck'\n        ],\n    ' ass ':\n        [\n            '[^a-z]ass ', '[^a-z]azz ', 'arrse', ' arse ', '@\\$\\$'\n                                                           '[^a-z]anus', ' a\\*s\\*s', '[^a-z]ass[^a-z ]',\n            'a[@#\\$%\\^&\\*][@#\\$%\\^&\\*]', '[^a-z]anal ', 'a s s'\n        ],\n    ' ass hole ':\n        [\n            ' a[s|z]*wipe', 'a[s|z]*[w]*h[o|0]+[l]*e', '@\\$\\$hole'\n        ],\n    ' bitch ':\n        [\n            'b[w]*i[t]*ch', 'b!tch',\n            'bi\\+ch', 'b!\\+ch', '(b)([^a-z]*)(i)([^a-z]*)(t)([^a-z]*)(c)([^a-z]*)(h)',\n            'biatch', 'bi\\*\\*h', 'bytch', 'b i t c h'\n        ],\n    ' bastard ':\n        [\n            'ba[s|z]+t[e|a]+rd'\n        ],\n    ' trans gender':\n        [\n            'transgender'\n        ],\n    ' gay ':\n        [\n            'gay'\n        ],\n    ' cock ':\n        [\n            '[^a-z]cock', 'c0ck', '[^a-z]cok ', 'c0k', '[^a-z]cok[^aeiou]', ' cawk',\n            '(c)([^a-z ])(o)([^a-z ]*)(c)([^a-z ]*)(k)', 'c o c k'\n        ],\n    ' dick ':\n        [\n            ' dick[^aeiou]', 'deek', 'd i c k'\n        ],\n    ' suck ':\n        [\n            'sucker', '(s)([^a-z ]*)(u)([^a-z ]*)(c)([^a-z ]*)(k)', 'sucks', '5uck', 's u c k'\n        ],\n    ' cunt ':\n        [\n            'cunt', 'c u n t'\n        ],\n    ' bull shit ':\n        [\n            'bullsh\\*t', 'bull\\$hit'\n        ],\n    ' homo sex ual':\n        [\n            'homosexual'\n        ],\n    ' jerk ':\n        [\n            'jerk'\n        ],\n    ' idiot ':\n        [\n            'i[d]+io[t]+', '(i)([^a-z ]*)(d)([^a-z ]*)(i)([^a-z ]*)(o)([^a-z ]*)(t)', 'idiots'\n                                                                                      'i d i o t'\n        ],\n    ' dumb ':\n        [\n            '(d)([^a-z ]*)(u)([^a-z ]*)(m)([^a-z ]*)(b)'\n        ],\n    ' shit ':\n        [\n            'shitty', '(s)([^a-z ]*)(h)([^a-z ]*)(i)([^a-z ]*)(t)', 'shite', '\\$hit', 's h i t'\n        ],\n    ' shit hole ':\n        [\n            'shythole'\n        ],\n    ' retard ':\n        [\n            'returd', 'retad', 'retard', 'wiktard', 'wikitud'\n        ],\n    ' rape ':\n        [\n            ' raped'\n        ],\n    ' dumb ass':\n        [\n            'dumbass', 'dubass'\n        ],\n    ' ass head':\n        [\n            'butthead'\n        ],\n    ' sex ':\n        [\n            'sexy', 's3x', 'sexuality'\n        ],\n    ' nigger ':\n        [\n            'nigger', 'ni[g]+a', ' nigr ', 'negrito', 'niguh', 'n3gr', 'n i g g e r'\n        ],\n    ' shut the fuck up':\n        [\n            'stfu'\n        ],\n    ' pussy ':\n        [\n            'pussy[^c]', 'pusy', 'pussi[^l]', 'pusses'\n        ],\n    ' faggot ':\n        [\n            'faggot', ' fa[g]+[s]*[^a-z ]', 'fagot', 'f a g g o t', 'faggit',\n            '(f)([^a-z ]*)(a)([^a-z ]*)([g]+)([^a-z ]*)(o)([^a-z ]*)(t)', 'fau[g]+ot', 'fae[g]+ot',\n        ],\n    ' mother fucker':\n        [\n            ' motha ', ' motha f', ' mother f', 'motherucker',\n        ],\n    ' whore ':\n        [\n            'wh\\*\\*\\*', 'w h o r e'\n        ],\n}\n\n\nclass PatternTokenizer(BaseTokenizer):\n    def __init__(self, lower=True, initial_filters=r\"[^a-z0-9!@#\\$%\\^\\&\\*_\\-,\\.' ]\", patterns=RE_PATTERNS,\n                 remove_repetitions=True):\n        self.lower = lower\n        self.patterns = patterns\n        self.initial_filters = initial_filters\n        self.remove_repetitions = remove_repetitions\n\n    def process_text(self, text):\n        x = self._preprocess(text)\n        for target, patterns in self.patterns.items():\n            for pat in patterns:\n                x = re.sub(pat, target, x)\n        x = re.sub(r\"[^a-z' ]\", ' ', x)\n        return x.split()\n\n    def process_ds(self, ds):\n        ### ds = Data series\n\n        # lower\n        ds = copy.deepcopy(ds)\n        if self.lower:\n            ds = ds.str.lower()\n        # remove special chars\n        if self.initial_filters is not None:\n            ds = ds.str.replace(self.initial_filters, ' ')\n        # fuuuuck => fuck\n        if self.remove_repetitions:\n            pattern = re.compile(r\"(.)\\1{2,}\", re.DOTALL) \n            ds = ds.str.replace(pattern, r\"\\1\")\n\n        for target, patterns in self.patterns.items():\n            for pat in patterns:\n                ds = ds.str.replace(pat, target)\n\n        ds = ds.str.replace(r\"[^a-z' ]\", ' ')\n\n        return ds.str.split()\n\n    def _preprocess(self, text):\n        # lower\n        if self.lower:\n            text = text.lower()\n\n        # remove special chars\n        if self.initial_filters is not None:\n            text = re.sub(self.initial_filters, ' ', text)\n\n        # fuuuuck => fuck\n        if self.remove_repetitions:\n            pattern = re.compile(r\"(.)\\1{2,}\", re.DOTALL)\n            text = pattern.sub(r\"\\1\", text)\n        return text","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = PatternTokenizer()\nfinal[\"comment_text_processed\"] = tokenizer.process_ds(final[\"comment_text\"]).str.join(sep=\" \")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm *.csv","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:34.242396Z","iopub.execute_input":"2021-12-02T03:27:34.2426Z","iopub.status.idle":"2021-12-02T03:27:34.246924Z","shell.execute_reply.started":"2021-12-02T03:27:34.242575Z","shell.execute_reply":"2021-12-02T03:27:34.245816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final.to_csv('all_in_one_jigsaw.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-02T03:27:34.248081Z","iopub.execute_input":"2021-12-02T03:27:34.248307Z","iopub.status.idle":"2021-12-02T03:27:59.552738Z","shell.execute_reply.started":"2021-12-02T03:27:34.248282Z","shell.execute_reply":"2021-12-02T03:27:59.551666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<p style = \"font-size:25px; \nfont-family: Helvetica; \nfont-weight : normal; \nbackground-color: #036EB7; \ncolor : #FFFFFF; \ntext-align: right; \npadding: 0px 15px; \nborder-radius:3px\">\n    Final!!!\n</p>","metadata":{}}]}