{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T06:53:07.668433Z","iopub.execute_input":"2022-08-13T06:53:07.668839Z","iopub.status.idle":"2022-08-13T06:53:07.676150Z","shell.execute_reply.started":"2022-08-13T06:53:07.668807Z","shell.execute_reply":"2022-08-13T06:53:07.675006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Goal of This Notebook is to get the toxic sentences and format them into a dataframe with 2 columns one having the comment and the other having a list of identites it targets**","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/jigsaw-unintended-bias-in-toxicity-classification/all_data.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T06:53:07.949483Z","iopub.execute_input":"2022-08-13T06:53:07.950249Z","iopub.status.idle":"2022-08-13T06:53:27.793928Z","shell.execute_reply.started":"2022-08-13T06:53:07.950214Z","shell.execute_reply":"2022-08-13T06:53:27.793057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_jigsaw = df.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T06:53:27.795600Z","iopub.execute_input":"2022-08-13T06:53:27.796113Z","iopub.status.idle":"2022-08-13T06:53:28.212735Z","shell.execute_reply.started":"2022-08-13T06:53:27.796082Z","shell.execute_reply":"2022-08-13T06:53:28.211527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_jigsaw = df_jigsaw.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T06:53:28.214096Z","iopub.execute_input":"2022-08-13T06:53:28.214463Z","iopub.status.idle":"2022-08-13T06:53:29.129958Z","shell.execute_reply.started":"2022-08-13T06:53:28.214430Z","shell.execute_reply":"2022-08-13T06:53:29.128756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_jigsaw.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-13T06:53:29.133231Z","iopub.execute_input":"2022-08-13T06:53:29.133800Z","iopub.status.idle":"2022-08-13T06:53:29.141835Z","shell.execute_reply.started":"2022-08-13T06:53:29.133697Z","shell.execute_reply":"2022-08-13T06:53:29.140659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = ['funny', 'wow', 'sad', 'likes',\n       'disagree', 'toxicity', 'severe_toxicity', 'obscene', 'sexual_explicit',\n       'identity_attack', 'insult', 'threat', 'male', 'female', 'transgender',\n       'other_gender', 'heterosexual', 'homosexual_gay_or_lesbian', 'bisexual',\n       'other_sexual_orientation', 'christian', 'jewish', 'muslim', 'hindu',\n       'buddhist', 'atheist', 'other_religion', 'black', 'white', 'asian',\n       'latino', 'other_race_or_ethnicity', 'physical_disability',\n       'intellectual_or_learning_disability', 'psychiatric_or_mental_illness',\n       'other_disability']","metadata":{"execution":{"iopub.status.busy":"2022-08-13T06:53:29.143399Z","iopub.execute_input":"2022-08-13T06:53:29.143729Z","iopub.status.idle":"2022-08-13T06:53:29.153936Z","shell.execute_reply.started":"2022-08-13T06:53:29.143699Z","shell.execute_reply":"2022-08-13T06:53:29.152929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in num_cols:\n    df_jigsaw[i] = df_jigsaw[i].apply(lambda x: 0 if x < 0.5 else 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T06:53:29.155250Z","iopub.execute_input":"2022-08-13T06:53:29.156495Z","iopub.status.idle":"2022-08-13T06:53:33.318275Z","shell.execute_reply.started":"2022-08-13T06:53:29.156458Z","shell.execute_reply":"2022-08-13T06:53:33.316804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_jigsaw","metadata":{"execution":{"iopub.status.busy":"2022-08-13T06:53:33.320079Z","iopub.execute_input":"2022-08-13T06:53:33.320892Z","iopub.status.idle":"2022-08-13T06:53:33.504960Z","shell.execute_reply.started":"2022-08-13T06:53:33.320843Z","shell.execute_reply":"2022-08-13T06:53:33.503807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_jigsaw = df_jigsaw[df_jigsaw['toxicity'] == 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T06:53:33.506575Z","iopub.execute_input":"2022-08-13T06:53:33.507283Z","iopub.status.idle":"2022-08-13T06:53:33.579783Z","shell.execute_reply.started":"2022-08-13T06:53:33.507240Z","shell.execute_reply":"2022-08-13T06:53:33.578526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nresult = df_jigsaw.to_json(orient=\"records\")\nparsed = json.loads(result)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T06:53:33.581634Z","iopub.execute_input":"2022-08-13T06:53:33.582004Z","iopub.status.idle":"2022-08-13T06:53:36.149731Z","shell.execute_reply.started":"2022-08-13T06:53:33.581970Z","shell.execute_reply":"2022-08-13T06:53:36.148577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls_identities = ['male','female','transgender','other_gender','heterosexual','homosexual_gay_or_lesbian','bisexual',\n                 'other_sexual_orientation','christian','jewish','muslim','hindu','buddhist','atheist','other_religion',\n                'black','white','asian','latino','other_race_or_ethnicity','physical_disability',\n                 'intellectual_or_learning_disability','psychiatric_or_mental_illness','other_disability'\n                ]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:01:14.796080Z","iopub.execute_input":"2022-08-13T07:01:14.797274Z","iopub.status.idle":"2022-08-13T07:01:14.803466Z","shell.execute_reply.started":"2022-08-13T07:01:14.797229Z","shell.execute_reply":"2022-08-13T07:01:14.802411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parsed[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T06:56:40.217710Z","iopub.execute_input":"2022-08-13T06:56:40.218191Z","iopub.status.idle":"2022-08-13T06:56:40.227015Z","shell.execute_reply.started":"2022-08-13T06:56:40.218156Z","shell.execute_reply":"2022-08-13T06:56:40.226041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls_comment_with_labels = []\nfor i in parsed:\n    comment_text = i['comment_text']\n    targets = []\n    for identity in ls_identities:\n        if i[identity] == 1:\n            targets.append(identity)\n    if targets == []:\n        continue\n    ls_comment_with_labels.append((comment_text, targets))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:04:18.654796Z","iopub.execute_input":"2022-08-13T07:04:18.656027Z","iopub.status.idle":"2022-08-13T07:04:19.084091Z","shell.execute_reply.started":"2022-08-13T07:04:18.655966Z","shell.execute_reply":"2022-08-13T07:04:19.083039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls_comment_with_labels[:10]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:04:36.833877Z","iopub.execute_input":"2022-08-13T07:04:36.834355Z","iopub.status.idle":"2022-08-13T07:04:36.843258Z","shell.execute_reply.started":"2022-08-13T07:04:36.834295Z","shell.execute_reply":"2022-08-13T07:04:36.842312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_processed = pd.DataFrame(ls_comment_with_labels, \n             columns=['comment', \n                      'targets'])","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:06:23.673957Z","iopub.execute_input":"2022-08-13T07:06:23.674671Z","iopub.status.idle":"2022-08-13T07:06:23.695722Z","shell.execute_reply.started":"2022-08-13T07:06:23.674633Z","shell.execute_reply":"2022-08-13T07:06:23.694096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_processed","metadata":{"execution":{"iopub.status.busy":"2022-08-13T07:06:26.426756Z","iopub.execute_input":"2022-08-13T07:06:26.427212Z","iopub.status.idle":"2022-08-13T07:06:26.446208Z","shell.execute_reply.started":"2022-08-13T07:06:26.427179Z","shell.execute_reply":"2022-08-13T07:06:26.445369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}