{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# MÔ TẢ\nDùng mô hình học máy dự đoán khả năng đoạn văn bản có mang tính toxic hay không?\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-05-26T18:25:45.367409Z","iopub.execute_input":"2021-05-26T18:25:45.367810Z","iopub.status.idle":"2021-05-26T18:25:45.390603Z","shell.execute_reply.started":"2021-05-26T18:25:45.367777Z","shell.execute_reply":"2021-05-26T18:25:45.389208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NHẬP DỮ LIỆU\n","metadata":{}},{"cell_type":"markdown","source":"** Khởi chạy thử data cần train**","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('../input/vankhang/jigsaw-toxic-comment-train.csv')\ndel(df_train['id'])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:42:15.370493Z","iopub.execute_input":"2021-05-26T18:42:15.371201Z","iopub.status.idle":"2021-05-26T18:42:18.799763Z","shell.execute_reply.started":"2021-05-26T18:42:15.371148Z","shell.execute_reply":"2021-05-26T18:42:18.798436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"** Sử dụng trình biên dịch tất cả về tiếng anh**","metadata":{}},{"cell_type":"code","source":"df_valid = pd.read_csv('../input/jigsaw-multilingual-toxic-test-translated/jigsaw_miltilingual_valid_translated.csv')\ndel(df_valid['id'])\ndf_valid.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:31:11.324509Z","iopub.execute_input":"2021-05-26T18:31:11.325064Z","iopub.status.idle":"2021-05-26T18:31:11.666896Z","shell.execute_reply.started":"2021-05-26T18:31:11.325016Z","shell.execute_reply":"2021-05-26T18:31:11.666086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Sử dụng trình biên dịch và test**","metadata":{}},{"cell_type":"code","source":"df_test = pd.read_csv('../input/jigsaw-multilingual-toxic-test-translated/jigsaw_miltilingual_test_translated.csv')\ndel(df_test['id'])\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:31:21.637912Z","iopub.execute_input":"2021-05-26T18:31:21.638321Z","iopub.status.idle":"2021-05-26T18:31:24.273412Z","shell.execute_reply.started":"2021-05-26T18:31:21.638290Z","shell.execute_reply":"2021-05-26T18:31:24.272441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**In ra số phần trăm toxic comments ở data cần train**","metadata":{}},{"cell_type":"code","source":"print(\"There are %.2f%% toxic comments in the training data.\"%(df_train['toxic'].value_counts()[1]/df_train['toxic'].value_counts()[0]*100))","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:43:12.424202Z","iopub.execute_input":"2021-05-26T18:43:12.424690Z","iopub.status.idle":"2021-05-26T18:43:12.439726Z","shell.execute_reply.started":"2021-05-26T18:43:12.424645Z","shell.execute_reply":"2021-05-26T18:43:12.437847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**In ra số phần trăm toxic comments khi thẩm định**","metadata":{}},{"cell_type":"code","source":"print(\"There are %.2f%% toxic comments in the validation data.\"%(df_valid['toxic'].value_counts()[1]/df_valid['toxic'].value_counts()[0]*100))","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:43:15.120879Z","iopub.execute_input":"2021-05-26T18:43:15.121365Z","iopub.status.idle":"2021-05-26T18:43:15.132714Z","shell.execute_reply.started":"2021-05-26T18:43:15.121329Z","shell.execute_reply":"2021-05-26T18:43:15.131190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Tỉ lệ toxic comments khi thẩm định so với data cần trian**","metadata":{}},{"cell_type":"code","source":"print(\"The validation dataframe represents a %.2f%% of the training data.\" %(df_valid.shape[0]/(df_train.shape[0]+df_valid.shape[0])))","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:43:17.066673Z","iopub.execute_input":"2021-05-26T18:43:17.067117Z","iopub.status.idle":"2021-05-26T18:43:17.073662Z","shell.execute_reply.started":"2021-05-26T18:43:17.067076Z","shell.execute_reply":"2021-05-26T18:43:17.072277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# PHÂN TÍCH DỮ LIỆU","metadata":{}},{"cell_type":"markdown","source":"**HUẤN LUYỆN**","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\nimport plotly.graph_objects as go\n\ncols = [col for col in df_train.columns]\ncols.remove('comment_text')\ntoxic_cats = {}\n\nfor i in cols:\n    i1 = i.capitalize()\n    i1 = i1.replace(\"_\", \" \")\n    toxic_cats[i1] = df_train[i].value_counts()[1]\n\n\n\n\n\nfig = px.bar(x=toxic_cats.values(), y=toxic_cats.keys(), text=toxic_cats.values(),\n             width=700, height=400, title='Nº of comments per toxicity level',\n             color=toxic_cats.values(),\n             labels={'x': 'Nº of comments', 'y': 'Level'})\nfig.update_layout(barmode='stack', yaxis={'categoryorder':'total ascending'})\n\nwith_toxic = {}\n\nfor i in cols:\n    i1 = i.capitalize()\n    i1 = i1.replace(\"_\", \" \")\n    with_toxic[i1] = sum(np.where((df_train['toxic'] == df_train[i]) & (df_train['toxic'] == 1),\n                                   True, False))\n\nfig = px.bar(x=with_toxic.values(), y=with_toxic.keys(), text=with_toxic.values(),\n             width=700, height=400, title='Nº of comments per toxicity level',\n             color=with_toxic.values(),\n             labels={'x': 'Nº of comments', 'y': 'Level'})\nfig.update_layout(barmode='stack', yaxis={'categoryorder':'total ascending'})\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:43:19.475063Z","iopub.execute_input":"2021-05-26T18:43:19.475483Z","iopub.status.idle":"2021-05-26T18:43:24.788379Z","shell.execute_reply.started":"2021-05-26T18:43:19.475450Z","shell.execute_reply":"2021-05-26T18:43:24.785483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.pie(values=toxic_cats.values(), names=toxic_cats.keys(), width=700, height=400,\n            title=\"Distribution of comments' toxicity categories\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:43:28.999753Z","iopub.execute_input":"2021-05-26T18:43:29.000224Z","iopub.status.idle":"2021-05-26T18:43:29.063704Z","shell.execute_reply.started":"2021-05-26T18:43:29.000175Z","shell.execute_reply":"2021-05-26T18:43:29.062365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(data=[\n    go.Bar(y=[a for a in toxic_cats.values()], x=[a for a in toxic_cats.keys()],\n           name='Total', marker_color='purple'),\n    go.Bar(y=[a for a in with_toxic.values()], x=[a for a in with_toxic.keys()],\n          name='Toxic as well', marker_color='yellow')\n])\n\nfig.update_layout(title='Are comments in other categories in toxic as well?', barmode='group', xaxis={'categoryorder':'total descending'})\n\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:43:31.892156Z","iopub.execute_input":"2021-05-26T18:43:31.892593Z","iopub.status.idle":"2021-05-26T18:43:31.913738Z","shell.execute_reply.started":"2021-05-26T18:43:31.892553Z","shell.execute_reply":"2021-05-26T18:43:31.912431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Chúng ta có thể thấy rõ mối quan hệ giữa các loại độc hại và các loại khác, vì vậy chúng ta sẽ thay thế các nhận xét được phân loại là không độc hại thành độc hại nếu chúng được đưa vào mức độ độc hại khác.**","metadata":{}},{"cell_type":"code","source":"toxic_bfr = df_train.toxic.value_counts()[1]\n\nfor i in range(len(df_train)):\n    if (df_train.loc[i,'toxic'] == 0) and (df_train.loc[i,'obscene'] == 1 or\n                                         df_train.loc[i, 'severe_toxic'] == 1 or\n                                         df_train.loc[i, 'threat'] == 1 or\n                                         df_train.loc[i, 'insult'] == 1 or\n                                         df_train.loc[i, 'identity_hate'] == 1):\n        df_train.loc[i,'toxic'] = 1\n        \ntoxic_after = df_train.toxic.value_counts()[1]\ntoxic_comments = toxic_after - toxic_bfr\nprint('There are %i new toxic comments.' %toxic_comments)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:53:55.825601Z","iopub.execute_input":"2021-05-26T18:53:55.826080Z","iopub.status.idle":"2021-05-26T18:53:55.919948Z","shell.execute_reply.started":"2021-05-26T18:53:55.826031Z","shell.execute_reply":"2021-05-26T18:53:55.917778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\n\ndel(df_train['obscene'])\ndel(df_train['identity_hate'])\ndel(df_train['insult'])\ndel(df_train['threat'])\ndel(df_train['severe_toxic'])\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:44:19.738577Z","iopub.execute_input":"2021-05-26T18:44:19.738999Z","iopub.status.idle":"2021-05-26T18:44:19.827603Z","shell.execute_reply.started":"2021-05-26T18:44:19.738951Z","shell.execute_reply":"2021-05-26T18:44:19.825536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# THẨM ĐỊNH","metadata":{}},{"cell_type":"code","source":"languages_val = {a:b for a,b in zip(df_valid['lang'].unique(), df_valid['lang'].value_counts())}\nlanguages_val['Spanish'] = languages_val.pop('es')\nlanguages_val['Italian'] = languages_val.pop('it')\nlanguages_val['Turkish'] = languages_val.pop('tr')\n\n\nfig = px.pie(values=languages_val.values(), names=languages_val.keys(), width=700, height=400,\n            title=\"Distribution of comments' languages in validation data\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:45:53.229108Z","iopub.execute_input":"2021-05-26T18:45:53.229571Z","iopub.status.idle":"2021-05-26T18:45:53.320293Z","shell.execute_reply.started":"2021-05-26T18:45:53.229526Z","shell.execute_reply":"2021-05-26T18:45:53.318523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TEST**","metadata":{}},{"cell_type":"code","source":"languages_test = {a:b for a,b in zip(df_test['lang'].unique(), df_test['lang'].value_counts())}\nlanguages_test['Spanish'] = languages_test.pop('es')\nlanguages_test['Italian'] = languages_test.pop('it')\nlanguages_test['Turkish'] = languages_test.pop('tr')\nlanguages_test['Russian'] = languages_test.pop('ru')\nlanguages_test['French'] = languages_test.pop('fr')\nlanguages_test['Portuguese'] = languages_test.pop('pt')\n\nfig = px.pie(values=languages_test.values(), names=languages_test.keys(), width=700, height=400,\n            title=\"Distribution of comments' languages in testing data\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:45:56.682675Z","iopub.execute_input":"2021-05-26T18:45:56.683097Z","iopub.status.idle":"2021-05-26T18:45:56.763305Z","shell.execute_reply.started":"2021-05-26T18:45:56.683060Z","shell.execute_reply":"2021-05-26T18:45:56.761940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XỬ LÝ DỮ LIỆU","metadata":{}},{"cell_type":"markdown","source":"**Dữ liệu được nhập vào là dữ liệu đã được dịch sang tiếng anh ở trên**","metadata":{}},{"cell_type":"code","source":"del(df_valid['lang'])\ndel(df_valid['comment_text'])\ndf_valid = df_valid.rename(columns={'translated':'comment_text'})\n\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:46:03.624057Z","iopub.execute_input":"2021-05-26T18:46:03.624520Z","iopub.status.idle":"2021-05-26T18:46:03.713949Z","shell.execute_reply.started":"2021-05-26T18:46:03.624471Z","shell.execute_reply":"2021-05-26T18:46:03.712009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([df_train, df_valid], ignore_index=True, axis=0)\n\ndf","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:46:09.388066Z","iopub.execute_input":"2021-05-26T18:46:09.388545Z","iopub.status.idle":"2021-05-26T18:46:09.412111Z","shell.execute_reply.started":"2021-05-26T18:46:09.388502Z","shell.execute_reply":"2021-05-26T18:46:09.410943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_toxic = df[df['toxic'] == 1]\ndf_toxic","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:46:26.344460Z","iopub.execute_input":"2021-05-26T18:46:26.345190Z","iopub.status.idle":"2021-05-26T18:46:26.363965Z","shell.execute_reply.started":"2021-05-26T18:46:26.345148Z","shell.execute_reply":"2021-05-26T18:46:26.362616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = df['comment_text']\ny = df['toxic']\n\nx_train, x_valid, y_train, y_valid = train_test_split(X, y,\n                                                       random_state=1,\n                                                       train_size=0.2\n                                                      )","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:46:46.018681Z","iopub.execute_input":"2021-05-26T18:46:46.019163Z","iopub.status.idle":"2021-05-26T18:46:46.077246Z","shell.execute_reply.started":"2021-05-26T18:46:46.019115Z","shell.execute_reply":"2021-05-26T18:46:46.076069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.sum()/y_train.shape[0]","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:46:50.510605Z","iopub.execute_input":"2021-05-26T18:46:50.511042Z","iopub.status.idle":"2021-05-26T18:46:50.519145Z","shell.execute_reply.started":"2021-05-26T18:46:50.510979Z","shell.execute_reply":"2021-05-26T18:46:50.517522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\nvec = TfidfVectorizer(decode_error='ignore',stop_words='english', max_df=0.8, max_features=3400)\nx_train = vec.fit_transform(x_train).todense()\nx_train = pd.DataFrame(x_train, columns=vec.get_feature_names())","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:47:00.890843Z","iopub.execute_input":"2021-05-26T18:47:00.891474Z","iopub.status.idle":"2021-05-26T18:47:00.989576Z","shell.execute_reply.started":"2021-05-26T18:47:00.891421Z","shell.execute_reply":"2021-05-26T18:47:00.988626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_valid = vec.transform(x_valid).todense()\nx_valid = pd.DataFrame(x_valid, columns=vec.get_feature_names())","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:48:05.229138Z","iopub.execute_input":"2021-05-26T18:48:05.230465Z","iopub.status.idle":"2021-05-26T18:48:05.435008Z","shell.execute_reply.started":"2021-05-26T18:48:05.230371Z","shell.execute_reply":"2021-05-26T18:48:05.434008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del(df)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:48:10.391068Z","iopub.execute_input":"2021-05-26T18:48:10.391498Z","iopub.status.idle":"2021-05-26T18:48:10.427611Z","shell.execute_reply.started":"2021-05-26T18:48:10.391464Z","shell.execute_reply":"2021-05-26T18:48:10.425785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\n\ndir(XGBClassifier)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:48:12.794651Z","iopub.execute_input":"2021-05-26T18:48:12.795165Z","iopub.status.idle":"2021-05-26T18:48:12.804308Z","shell.execute_reply.started":"2021-05-26T18:48:12.795115Z","shell.execute_reply":"2021-05-26T18:48:12.802999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learning_rate = 0.1; max_depth=6; colsample_bytree=1;gamma=1; n_jobs=4","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:48:20.396095Z","iopub.execute_input":"2021-05-26T18:48:20.396519Z","iopub.status.idle":"2021-05-26T18:48:20.403190Z","shell.execute_reply.started":"2021-05-26T18:48:20.396481Z","shell.execute_reply":"2021-05-26T18:48:20.401969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score\n\nmodel = XGBClassifier(n_estimators=585,\n                      use_label_encoder=False,\n                      learning_rate=0.1,\n                      max_depth=6,\n                      colsample_bytree=1,\n                      gamma=1,\n                      n_jobs=4,\n                      scale_pos_weight=10000,\n                      random_state=1)\n\nmodel.fit(x_train, y_train)\n\npreds = model.predict(x_valid)\n\nf1 = f1_score(preds, y_valid)\n\nprint(\"F1 Score: %.4f\" %f1)\n#0.6852","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:48:22.924778Z","iopub.execute_input":"2021-05-26T18:48:22.925415Z","iopub.status.idle":"2021-05-26T18:48:23.368977Z","shell.execute_reply.started":"2021-05-26T18:48:22.925360Z","shell.execute_reply":"2021-05-26T18:48:23.367206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\n\ncm = confusion_matrix(y_valid, preds, labels=model.classes_, normalize='true')\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm,\n                             display_labels=model.classes_)\ndisp.plot()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:48:31.722509Z","iopub.execute_input":"2021-05-26T18:48:31.722959Z","iopub.status.idle":"2021-05-26T18:48:31.749769Z","shell.execute_reply.started":"2021-05-26T18:48:31.722922Z","shell.execute_reply":"2021-05-26T18:48:31.748099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.best_ntree_limit","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KIỂM TRA DỮ LIỆU","metadata":{}},{"cell_type":"code","source":"x_test = vec.transform(df_test['translated']).todense()\nx_test = pd.DataFrame(x_test, columns=vec.get_feature_names())\n#x_test = pca.transform(x_test)","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:48:36.099979Z","iopub.execute_input":"2021-05-26T18:48:36.100408Z","iopub.status.idle":"2021-05-26T18:48:43.784663Z","shell.execute_reply.started":"2021-05-26T18:48:36.100376Z","shell.execute_reply":"2021-05-26T18:48:43.783674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_test = model.predict(x_test)\n\n# Save test predictions to file\noutput = pd.DataFrame({'id': df_test.index,\n                       'toxic': preds_test})\noutput.to_csv('submission.csv', index=False)\n\noutput.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:48:47.264219Z","iopub.execute_input":"2021-05-26T18:48:47.264618Z","iopub.status.idle":"2021-05-26T18:48:51.775673Z","shell.execute_reply.started":"2021-05-26T18:48:47.264586Z","shell.execute_reply":"2021-05-26T18:48:51.773054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output['toxic'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-05-26T18:48:56.861094Z","iopub.execute_input":"2021-05-26T18:48:56.861504Z","iopub.status.idle":"2021-05-26T18:48:56.883601Z","shell.execute_reply.started":"2021-05-26T18:48:56.861470Z","shell.execute_reply":"2021-05-26T18:48:56.882340Z"},"trusted":true},"execution_count":null,"outputs":[]}]}