{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n-------\nNowadays, it's difficult to chat online without reading some toxic comments. To solve this problem, an option is to social medias forbid this harmful content. One way to do so is to create machine learning algorithms that can tell us if a comment is or isn't toxic. That's our goal in this Notebook.\n\nGiven a training dataset, we will build a model to predict if a english comment is toxic. But there is another problem - the testing dataset has comments in differents languages. A solution is to translate them to English, and that's what https://www.kaggle.com/kashnitsky has already done for us using the Yandex.Translate's API. We will analyse the data, preprocess the comments and use a EDA Model to reach our goal. So, let's start! Please let me know if you liked the notebook and please comment below how can I improve it! ","metadata":{}},{"cell_type":"markdown","source":"# Importing Data\n-------","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\n\ndf_train = pd.read_csv('../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\ndel(df_train['id'])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:05.358957Z","iopub.execute_input":"2021-07-17T19:34:05.359376Z","iopub.status.idle":"2021-07-17T19:34:08.319498Z","shell.execute_reply.started":"2021-07-17T19:34:05.359283Z","shell.execute_reply":"2021-07-17T19:34:08.318453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_valid = pd.read_csv('../input/jigsaw-multilingual-toxic-test-translated/jigsaw_miltilingual_valid_translated.csv')\ndel(df_valid['id'])\ndf_valid.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:08.320857Z","iopub.execute_input":"2021-07-17T19:34:08.321149Z","iopub.status.idle":"2021-07-17T19:34:08.567793Z","shell.execute_reply.started":"2021-07-17T19:34:08.32112Z","shell.execute_reply":"2021-07-17T19:34:08.566786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('../input/jigsaw-multilingual-toxic-test-translated/jigsaw_miltilingual_test_translated.csv')\ndel(df_test['id'])\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:08.569475Z","iopub.execute_input":"2021-07-17T19:34:08.569776Z","iopub.status.idle":"2021-07-17T19:34:10.318633Z","shell.execute_reply.started":"2021-07-17T19:34:08.569748Z","shell.execute_reply":"2021-07-17T19:34:10.317571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Analysis\n------------------","metadata":{}},{"cell_type":"markdown","source":"## Train","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\nimport plotly.graph_objects as go\n\ncols = [col for col in df_train.columns]\ncols.remove('comment_text')\ntoxic_cats = {}\n\nfor i in cols:\n    i1 = i.capitalize()\n    i1 = i1.replace(\"_\", \" \")\n    toxic_cats[i1] = df_train[i].value_counts()[1]\n\n\n\n\n\nfig = px.bar(x=toxic_cats.values(), y=toxic_cats.keys(), text=toxic_cats.values(),\n             width=700, height=400, title='Nº of comments per toxicity level',\n             color=toxic_cats.values(),\n             labels={'x': 'Nº of comments', 'y': 'Level'})\nfig.update_layout(barmode='stack', yaxis={'categoryorder':'total ascending'})\n\nwith_toxic = {}\n\nfor i in cols:\n    i1 = i.capitalize()\n    i1 = i1.replace(\"_\", \" \")\n    with_toxic[i1] = sum(np.where((df_train['toxic'] == df_train[i]) & (df_train['toxic'] == 1),\n                                   True, False))\n\nfig = px.bar(x=with_toxic.values(), y=with_toxic.keys(), text=with_toxic.values(),\n             width=700, height=400, title='Nº of comments per toxicity level',\n             color=with_toxic.values(),\n             labels={'x': 'Nº of comments', 'y': 'Level'})\nfig.update_layout(barmode='stack', yaxis={'categoryorder':'total ascending'})\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:10.320475Z","iopub.execute_input":"2021-07-17T19:34:10.321057Z","iopub.status.idle":"2021-07-17T19:34:18.423294Z","shell.execute_reply.started":"2021-07-17T19:34:10.320999Z","shell.execute_reply":"2021-07-17T19:34:18.42207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.pie(values=toxic_cats.values(), names=toxic_cats.keys(), width=700, height=400,\n            title=\"Distribution of comments' toxicity categories\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:18.424825Z","iopub.execute_input":"2021-07-17T19:34:18.425111Z","iopub.status.idle":"2021-07-17T19:34:18.494843Z","shell.execute_reply.started":"2021-07-17T19:34:18.425084Z","shell.execute_reply":"2021-07-17T19:34:18.493809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(data=[\n    go.Bar(y=[a for a in toxic_cats.values()], x=[a for a in toxic_cats.keys()],\n           name='Total', marker_color='purple'),\n    go.Bar(y=[a for a in with_toxic.values()], x=[a for a in with_toxic.keys()],\n          name='Toxic as well', marker_color='yellow')\n])\n\nfig.update_layout(title='Are comments in other categories in toxic as well?', barmode='group', xaxis={'categoryorder':'total descending'})\n\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:18.496402Z","iopub.execute_input":"2021-07-17T19:34:18.497029Z","iopub.status.idle":"2021-07-17T19:34:18.518429Z","shell.execute_reply.started":"2021-07-17T19:34:18.496978Z","shell.execute_reply":"2021-07-17T19:34:18.517446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can clearly see the relation between toxic and other categories, so we will replace the comments that are classified as non-toxic to toxic if they are included in other level of toxicity.","metadata":{}},{"cell_type":"code","source":"toxic_bfr = df_train.toxic.value_counts()[1]\n\nfor i in range(len(df_train)):\n    if df_train.loc[i,'toxic'] == 0 and (df_train.loc[i, 'obscene'] == 1 or\n                                         df_train.loc[i, 'severe_toxic'] == 1 or\n                                         df_train.loc[i, 'threat'] == 1 or\n                                         df_train.loc[i, 'insult'] == 1 or\n                                         df_train.loc[i, 'identity_hate'] == 1):\n        df_train.loc[i,'toxic'] = 1\n        \ntoxic_after = df_train.toxic.value_counts()[1]\ntoxic_comments = toxic_after - toxic_bfr\nprint('There are %i new toxic comments.' %toxic_comments)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:18.519624Z","iopub.execute_input":"2021-07-17T19:34:18.519919Z","iopub.status.idle":"2021-07-17T19:34:37.211816Z","shell.execute_reply.started":"2021-07-17T19:34:18.519889Z","shell.execute_reply":"2021-07-17T19:34:37.210631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's a huge dataset, so we will delete some columns and dataframes to save RAM Memory (I've already allocated more memory that I can count to solve it, please let me know if you have some tips on how to save more RAM memory).","metadata":{}},{"cell_type":"code","source":"import gc \n\ndel(df_train['obscene'])\ndel(df_train['identity_hate'])\ndel(df_train['insult'])\ndel(df_train['threat'])\ndel(df_train['severe_toxic'])\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:37.215984Z","iopub.execute_input":"2021-07-17T19:34:37.216338Z","iopub.status.idle":"2021-07-17T19:34:37.355681Z","shell.execute_reply.started":"2021-07-17T19:34:37.216306Z","shell.execute_reply":"2021-07-17T19:34:37.354762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Validation","metadata":{}},{"cell_type":"code","source":"languages_val = {a:b for a,b in zip(df_valid['lang'].unique(), df_valid['lang'].value_counts())}\nlanguages_val['Spanish'] = languages_val.pop('es')\nlanguages_val['Italian'] = languages_val.pop('it')\nlanguages_val['Turkish'] = languages_val.pop('tr')\n\n\nfig = px.pie(values=languages_val.values(), names=languages_val.keys(), width=700, height=400,\n            title=\"Distribution of comments' languages in validation data\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:37.357447Z","iopub.execute_input":"2021-07-17T19:34:37.357735Z","iopub.status.idle":"2021-07-17T19:34:37.417584Z","shell.execute_reply.started":"2021-07-17T19:34:37.357707Z","shell.execute_reply":"2021-07-17T19:34:37.416284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test","metadata":{}},{"cell_type":"code","source":"languages_test = {a:b for a,b in zip(df_test['lang'].unique(), df_test['lang'].value_counts())}\nlanguages_test['Spanish'] = languages_test.pop('es')\nlanguages_test['Italian'] = languages_test.pop('it')\nlanguages_test['Turkish'] = languages_test.pop('tr')\nlanguages_test['Russian'] = languages_test.pop('ru')\nlanguages_test['French'] = languages_test.pop('fr')\nlanguages_test['Portuguese'] = languages_test.pop('pt')\n\nfig = px.pie(values=languages_test.values(), names=languages_test.keys(), width=700, height=400,\n            title=\"Distribution of comments' languages in testing data\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:37.419006Z","iopub.execute_input":"2021-07-17T19:34:37.419337Z","iopub.status.idle":"2021-07-17T19:34:37.49296Z","shell.execute_reply.started":"2021-07-17T19:34:37.419304Z","shell.execute_reply":"2021-07-17T19:34:37.492016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing\n-------","metadata":{}},{"cell_type":"markdown","source":"We already imported a dataset translated to english using Yandex.Translate, so we will use only the translated comments.","metadata":{}},{"cell_type":"code","source":"print(\"There are %.2f%% toxic comments in the training data.\"%(df_train['toxic'].value_counts()[1]/df_train['toxic'].value_counts()[0]*100))","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:37.494037Z","iopub.execute_input":"2021-07-17T19:34:37.494339Z","iopub.status.idle":"2021-07-17T19:34:37.507117Z","shell.execute_reply.started":"2021-07-17T19:34:37.49431Z","shell.execute_reply":"2021-07-17T19:34:37.506048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"There are %.2f%% toxic comments in the validation data.\"%(df_valid['toxic'].value_counts()[1]/df_valid['toxic'].value_counts()[0]*100))","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:37.508701Z","iopub.execute_input":"2021-07-17T19:34:37.509125Z","iopub.status.idle":"2021-07-17T19:34:37.518162Z","shell.execute_reply.started":"2021-07-17T19:34:37.509079Z","shell.execute_reply":"2021-07-17T19:34:37.51727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"The validation dataframe represents a %.2f%% of the training data.\" %(df_valid.shape[0]/(df_train.shape[0]+df_valid.shape[0])))","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:37.519267Z","iopub.execute_input":"2021-07-17T19:34:37.519545Z","iopub.status.idle":"2021-07-17T19:34:37.530855Z","shell.execute_reply.started":"2021-07-17T19:34:37.519517Z","shell.execute_reply":"2021-07-17T19:34:37.529834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our validation dataframe represents only 0.03% of training data, and the toxic comments are disproportionate distributed between both dataframes. So we will need to join them and split them randomly to have a more accurate result.","metadata":{}},{"cell_type":"code","source":"del(df_valid['lang'])\ndel(df_valid['comment_text'])\ndf_valid = df_valid.rename(columns={'translated':'comment_text'})\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:37.532096Z","iopub.execute_input":"2021-07-17T19:34:37.532482Z","iopub.status.idle":"2021-07-17T19:34:37.662963Z","shell.execute_reply.started":"2021-07-17T19:34:37.532446Z","shell.execute_reply":"2021-07-17T19:34:37.661885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([df_train, df_valid], ignore_index=True, axis=0)\n\ndf","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:37.664097Z","iopub.execute_input":"2021-07-17T19:34:37.664394Z","iopub.status.idle":"2021-07-17T19:34:37.685733Z","shell.execute_reply.started":"2021-07-17T19:34:37.664366Z","shell.execute_reply":"2021-07-17T19:34:37.684495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = df['comment_text']\ny = df['toxic']\n\nx_train, x_valid, y_train, y_valid = train_test_split(X, y,\n                                                       random_state=1,\n                                                       train_size=0.8\n                                                      )","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:37.687324Z","iopub.execute_input":"2021-07-17T19:34:37.687636Z","iopub.status.idle":"2021-07-17T19:34:38.632792Z","shell.execute_reply.started":"2021-07-17T19:34:37.687604Z","shell.execute_reply":"2021-07-17T19:34:38.631515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\nvec = TfidfVectorizer(decode_error='ignore',stop_words='english', max_df=0.8, max_features=1600)\nx_train = vec.fit_transform(x_train).todense()\nx_train = pd.DataFrame(x_train, columns=vec.get_feature_names())","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:38.634548Z","iopub.execute_input":"2021-07-17T19:34:38.63511Z","iopub.status.idle":"2021-07-17T19:34:55.450111Z","shell.execute_reply.started":"2021-07-17T19:34:38.634962Z","shell.execute_reply":"2021-07-17T19:34:55.449017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_valid = vec.transform(x_valid).todense()\nx_valid = pd.DataFrame(x_valid, columns=vec.get_feature_names())","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:55.451345Z","iopub.execute_input":"2021-07-17T19:34:55.451653Z","iopub.status.idle":"2021-07-17T19:34:59.408099Z","shell.execute_reply.started":"2021-07-17T19:34:55.451622Z","shell.execute_reply":"2021-07-17T19:34:59.407103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(df)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:59.409337Z","iopub.execute_input":"2021-07-17T19:34:59.409628Z","iopub.status.idle":"2021-07-17T19:34:59.414Z","shell.execute_reply.started":"2021-07-17T19:34:59.409601Z","shell.execute_reply":"2021-07-17T19:34:59.412904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"There are %.2f%% toxic comments in train data.\"%(y_train.sum()/len(y_train)*100))","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:59.415243Z","iopub.execute_input":"2021-07-17T19:34:59.415568Z","iopub.status.idle":"2021-07-17T19:34:59.430491Z","shell.execute_reply.started":"2021-07-17T19:34:59.415537Z","shell.execute_reply":"2021-07-17T19:34:59.42928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The training dataset is not well balanced (there are way more non-toxic comments than toxic ones). We will use SMOTE to add new toxic comments. ","metadata":{}},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\n\nsm = SMOTE(random_state=1)\n\nx_train, y_train = sm.fit_resample(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:34:59.433808Z","iopub.execute_input":"2021-07-17T19:34:59.434118Z","iopub.status.idle":"2021-07-17T19:35:43.242361Z","shell.execute_reply.started":"2021-07-17T19:34:59.434088Z","shell.execute_reply":"2021-07-17T19:35:43.241221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:35:43.243842Z","iopub.execute_input":"2021-07-17T19:35:43.244295Z","iopub.status.idle":"2021-07-17T19:35:43.250952Z","shell.execute_reply.started":"2021-07-17T19:35:43.244229Z","shell.execute_reply":"2021-07-17T19:35:43.249763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.tail()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T19:35:43.255622Z","iopub.execute_input":"2021-07-17T19:35:43.256109Z","iopub.status.idle":"2021-07-17T19:35:43.303383Z","shell.execute_reply.started":"2021-07-17T19:35:43.256062Z","shell.execute_reply":"2021-07-17T19:35:43.302398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training our Model\n-----------","metadata":{}},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score\n\nmodel = XGBClassifier(n_estimators=800,\n                      use_label_encoder=False,\n                      learning_rate=0.1,\n                      max_depth=6,\n                      gamma=1,\n                      scale_pos_weight=7,\n                      random_state=1)\n\nmodel.fit(x_train, y_train)\n\npreds = model.predict(x_valid)\n\nf1 = f1_score(preds, y_valid)\n\nprint(\"F1 Score: %.4f\" %f1)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-01T20:15:56.799828Z","iopub.execute_input":"2021-06-01T20:15:56.800181Z","iopub.status.idle":"2021-06-01T22:02:19.554097Z","shell.execute_reply.started":"2021-06-01T20:15:56.800151Z","shell.execute_reply":"2021-06-01T22:02:19.552789Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\ncm = confusion_matrix(y_valid, preds, labels=model.classes_, normalize='true')\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm,\n                             display_labels=model.classes_)\ndisp.plot() ","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:24:04.77422Z","iopub.status.idle":"2021-07-17T23:24:04.774786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predictions\n------","metadata":{}},{"cell_type":"code","source":"x_test = vec.transform(df_test['translated']).todense()\nx_test = pd.DataFrame(x_test, columns=vec.get_feature_names())","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:24:04.776204Z","iopub.status.idle":"2021-07-17T23:24:04.776905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_test = model.predict(x_test)\n\n# Save test predictions to file\noutput = pd.DataFrame({'id': df_test.index,\n                       'toxic': preds_test})\noutput.to_csv('submission.csv', index=False)\n\noutput.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:24:04.778186Z","iopub.status.idle":"2021-07-17T23:24:04.778857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output['toxic'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-07-17T23:24:04.78007Z","iopub.status.idle":"2021-07-17T23:24:04.780612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If you came this far, thank you so much. As I said before, please let me know if you liked it and tell me how can I improve this solution!","metadata":{}}]}