{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install pyicu","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:47:03.113584Z","iopub.execute_input":"2022-01-15T07:47:03.114021Z","iopub.status.idle":"2022-01-15T07:48:39.140905Z","shell.execute_reply.started":"2022-01-15T07:47:03.113986Z","shell.execute_reply":"2022-01-15T07:48:39.139135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install pycld2","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:48:39.144343Z","iopub.execute_input":"2022-01-15T07:48:39.144862Z","iopub.status.idle":"2022-01-15T07:49:19.630653Z","shell.execute_reply.started":"2022-01-15T07:48:39.144815Z","shell.execute_reply":"2022-01-15T07:49:19.629427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nfrom tqdm.notebook import tqdm\ntqdm.pandas()\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\n\nwarnings.filterwarnings(\"ignore\")\n\nimport nltk\nfrom nltk.corpus import wordnet, stopwords\nfrom nltk import *\nfrom wordcloud import WordCloud, STOPWORDS\nimport re\n\nimport sys\nfrom termcolor import colored\nfrom polyglot.detect import Detector\nfrom polyglot.utils import pretty_list","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:49:19.632839Z","iopub.execute_input":"2022-01-15T07:49:19.633156Z","iopub.status.idle":"2022-01-15T07:49:21.335726Z","shell.execute_reply.started":"2022-01-15T07:49:19.633110Z","shell.execute_reply":"2022-01-15T07:49:21.334824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Reading the training data file","metadata":{}},{"cell_type":"code","source":"train1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\ntrain2 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:49:21.338306Z","iopub.execute_input":"2022-01-15T07:49:21.338549Z","iopub.status.idle":"2022-01-15T07:49:47.640065Z","shell.execute_reply.started":"2022-01-15T07:49:21.338521Z","shell.execute_reply":"2022-01-15T07:49:47.639122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:49:47.641891Z","iopub.execute_input":"2022-01-15T07:49:47.642192Z","iopub.status.idle":"2022-01-15T07:49:47.666078Z","shell.execute_reply.started":"2022-01-15T07:49:47.642147Z","shell.execute_reply":"2022-01-15T07:49:47.665059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:49:47.667603Z","iopub.execute_input":"2022-01-15T07:49:47.668218Z","iopub.status.idle":"2022-01-15T07:49:47.704583Z","shell.execute_reply.started":"2022-01-15T07:49:47.668169Z","shell.execute_reply":"2022-01-15T07:49:47.703489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2.toxic = train2.toxic.round().astype(int)\ntrain = pd.concat([train1[['comment_text', 'toxic']],\n    train2[['comment_text', 'toxic']].query('toxic==1'),\n    train2[['comment_text', 'toxic']].query('toxic==0').sample(n=100000)\n    ])\n#rate=10\n#train = train[::rate]\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:49:47.706432Z","iopub.execute_input":"2022-01-15T07:49:47.706784Z","iopub.status.idle":"2022-01-15T07:49:48.434432Z","shell.execute_reply.started":"2022-01-15T07:49:47.706737Z","shell.execute_reply":"2022-01-15T07:49:48.433435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Descriptive Analysis","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:49:48.435875Z","iopub.execute_input":"2022-01-15T07:49:48.436548Z","iopub.status.idle":"2022-01-15T07:49:48.514094Z","shell.execute_reply.started":"2022-01-15T07:49:48.436502Z","shell.execute_reply":"2022-01-15T07:49:48.512815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:49:48.515892Z","iopub.execute_input":"2022-01-15T07:49:48.516378Z","iopub.status.idle":"2022-01-15T07:49:48.543366Z","shell.execute_reply.started":"2022-01-15T07:49:48.516317Z","shell.execute_reply":"2022-01-15T07:49:48.542222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Train data shape: {colored(train.shape, 'red', attrs=['bold'])}\")","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:49:48.546820Z","iopub.execute_input":"2022-01-15T07:49:48.547248Z","iopub.status.idle":"2022-01-15T07:49:48.552589Z","shell.execute_reply.started":"2022-01-15T07:49:48.547207Z","shell.execute_reply":"2022-01-15T07:49:48.551329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking for Null Values","metadata":{}},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:49:48.554732Z","iopub.execute_input":"2022-01-15T07:49:48.555576Z","iopub.status.idle":"2022-01-15T07:49:48.647259Z","shell.execute_reply.started":"2022-01-15T07:49:48.555468Z","shell.execute_reply":"2022-01-15T07:49:48.645902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis\n### Word Cloud Creation","metadata":{}},{"cell_type":"code","source":"def slashn(x):\n    if type(x) == str:\n        return x.replace(\"\\n\", \"\")\n    else:\n        return \"\"","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:49:48.650614Z","iopub.execute_input":"2022-01-15T07:49:48.651012Z","iopub.status.idle":"2022-01-15T07:49:48.658837Z","shell.execute_reply.started":"2022-01-15T07:49:48.650978Z","shell.execute_reply":"2022-01-15T07:49:48.657708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nltk.download('punkt') #tokenizer\nnltk.download('stopwords') #handle stopwords\nnltk.download('wordnet') #Lemmatization\n\nstop_words = stopwords.words('english')","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:50:10.468599Z","iopub.execute_input":"2022-01-15T07:50:10.468994Z","iopub.status.idle":"2022-01-15T07:50:10.751502Z","shell.execute_reply.started":"2022-01-15T07:50:10.468930Z","shell.execute_reply":"2022-01-15T07:50:10.750460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def word_count_review(docs):\n    text = ' '.join([slashn(abstract) for abstract in docs])\n    corpus = str(text.lower())\n    txt = re.sub(r'[^a-z0-9]+',' ',str(corpus)).strip()\n    tokens = word_tokenize(txt)\n    words = [t for t in tokens if t not in stop_words]\n    lemma = WordNetLemmatizer()\n    l = [lemma.lemmatize(w) for w in words]\n    fdq = FreqDist(l)\n    return fdq","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:50:13.940136Z","iopub.execute_input":"2022-01-15T07:50:13.940426Z","iopub.status.idle":"2022-01-15T07:50:13.948412Z","shell.execute_reply.started":"2022-01-15T07:50:13.940396Z","shell.execute_reply":"2022-01-15T07:50:13.947387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fd = word_count_review(train.comment_text)\nfd","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:50:19.653344Z","iopub.execute_input":"2022-01-15T07:50:19.654373Z","iopub.status.idle":"2022-01-15T07:55:36.903292Z","shell.execute_reply.started":"2022-01-15T07:50:19.654322Z","shell.execute_reply":"2022-01-15T07:55:36.902270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,10),dpi=150)\nwc = WordCloud(scale=10).generate_from_frequencies(fd)\n\nplt.imshow(wc)\nplt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:55:36.905554Z","iopub.execute_input":"2022-01-15T07:55:36.906044Z","iopub.status.idle":"2022-01-15T07:55:40.605113Z","shell.execute_reply.started":"2022-01-15T07:55:36.905998Z","shell.execute_reply":"2022-01-15T07:55:40.601162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Getting acquainted with POLYGLOT library\n#### Languages Supported","metadata":{}},{"cell_type":"code","source":"print(pretty_list(Detector.supported_languages()))","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:55:46.904007Z","iopub.execute_input":"2022-01-15T07:55:46.904318Z","iopub.status.idle":"2022-01-15T07:55:46.911423Z","shell.execute_reply.started":"2022-01-15T07:55:46.904282Z","shell.execute_reply":"2022-01-15T07:55:46.910204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_language(text):\n    return Detector(\"\".join(x for x in text if x.isprintable()),quiet=True).languages[0].name\nh = get_language(\"Helló, hogy vagy\")\ni = get_language(\"Dia duit, conas atá tú\")\ne = get_language(\"hello, how are you\")\np = get_language(\"ਹੈਲੋ ਤੁਸੀ ਕਿਵੇਂ ਹੋ\")\nt = get_language(\"сәлам, хәлләрең ничек\")\nk = get_language(\"안녕하세요. 어떻게 지내세요\")\nm = get_language(\"ഹലോ, നിങ്ങൾക്ക് സുഖമാണോ\")","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:55:49.275918Z","iopub.execute_input":"2022-01-15T07:55:49.276366Z","iopub.status.idle":"2022-01-15T07:55:49.290854Z","shell.execute_reply.started":"2022-01-15T07:55:49.276324Z","shell.execute_reply":"2022-01-15T07:55:49.289742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Helló, hogy vagy: {colored(h, 'blue', attrs=['bold','underline'])}\")\nprint(f\"Dia duit, conas atá tú: {colored(i, 'red', attrs=['bold','underline'])}\")\nprint(f\"hello, how are you: {colored(e, 'yellow', attrs=['bold','underline'])}\")\nprint(f\"ਹੈਲੋ ਤੁਸੀ ਕਿਵੇਂ ਹੋ: {colored(p, 'cyan', attrs=['bold','underline'])}\")\nprint(f\"сәлам, хәлләрең ничек: {colored(t, 'white', attrs=['bold','underline'])}\")\nprint(f\"안녕하세요. 어떻게 지내세요: {colored(k, 'magenta', attrs=['bold','underline'])}\")\nprint(f\"ഹലോ, നിങ്ങൾക്ക് സുഖമാണ: {colored(m, 'green', attrs=['bold','underline'])}\")","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:55:51.225535Z","iopub.execute_input":"2022-01-15T07:55:51.225807Z","iopub.status.idle":"2022-01-15T07:55:51.236505Z","shell.execute_reply.started":"2022-01-15T07:55:51.225779Z","shell.execute_reply":"2022-01-15T07:55:51.235012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Assigning a column of languages corresponding to the comment_text column","metadata":{}},{"cell_type":"code","source":"train['language'] = train[\"comment_text\"].apply(get_language)","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:55:53.579964Z","iopub.execute_input":"2022-01-15T07:55:53.580712Z","iopub.status.idle":"2022-01-15T07:56:32.576999Z","shell.execute_reply.started":"2022-01-15T07:55:53.580667Z","shell.execute_reply":"2022-01-15T07:56:32.576014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:58:03.673282Z","iopub.execute_input":"2022-01-15T07:58:03.673626Z","iopub.status.idle":"2022-01-15T07:58:03.686752Z","shell.execute_reply.started":"2022-01-15T07:58:03.673574Z","shell.execute_reply":"2022-01-15T07:58:03.685539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Languages in the training data set and their counts","metadata":{}},{"cell_type":"code","source":"train.language.unique()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:58:05.497631Z","iopub.execute_input":"2022-01-15T07:58:05.497934Z","iopub.status.idle":"2022-01-15T07:58:05.552437Z","shell.execute_reply.started":"2022-01-15T07:58:05.497903Z","shell.execute_reply":"2022-01-15T07:58:05.551519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of Unique Languages:\",train.language.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:58:08.201742Z","iopub.execute_input":"2022-01-15T07:58:08.202051Z","iopub.status.idle":"2022-01-15T07:58:08.258231Z","shell.execute_reply.started":"2022-01-15T07:58:08.201998Z","shell.execute_reply":"2022-01-15T07:58:08.257173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Comparison between English and Non English Languages","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\n\nlang_list = sorted(list(set(train[\"language\"])))\ncounts = [list(train[\"language\"]).count(cont) for cont in lang_list]\ndf = pd.DataFrame(np.transpose([lang_list, counts]))\ndf.columns = [\"Language\", \"Count\"]\ndf[\"Count\"] = df[\"Count\"].apply(int)\n\ndf_en = pd.DataFrame(np.transpose([[\"English\", \"Non-English\"], [max(counts), sum(counts) - max(counts)]]))\ndf_en.columns = [\"Language\", \"Count\"]\ndf_en.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:58:10.312606Z","iopub.execute_input":"2022-01-15T07:58:10.313072Z","iopub.status.idle":"2022-01-15T07:58:23.353414Z","shell.execute_reply.started":"2022-01-15T07:58:10.313036Z","shell.execute_reply":"2022-01-15T07:58:23.352389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_en.Count = df_en.Count.astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:59:14.880859Z","iopub.execute_input":"2022-01-15T07:59:14.881191Z","iopub.status.idle":"2022-01-15T07:59:14.887322Z","shell.execute_reply.started":"2022-01-15T07:59:14.881134Z","shell.execute_reply":"2022-01-15T07:59:14.885832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_en.plot.bar(x=\"Language\", y=\"Count\", rot=0)","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:59:18.196112Z","iopub.execute_input":"2022-01-15T07:59:18.196687Z","iopub.status.idle":"2022-01-15T07:59:18.446953Z","shell.execute_reply.started":"2022-01-15T07:59:18.196609Z","shell.execute_reply":"2022-01-15T07:59:18.445996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Comparison of Non English languages with comments appearing between 20 and 30 times","metadata":{}},{"cell_type":"code","source":"dfq = df.query(\"Language != 'English' and Language != 'un'\").query(\"Count >= 20 and Count <= 30\")\nfig1 = px.bar(dfq, y=\"Language\", x=\"Count\", title=\"Language of non-English comments\", text=\"Count\", orientation=\"h\",\n             pattern_shape=\"Language\", pattern_shape_sequence=[\"|\", \"/\", \"+\"], height=500)\nfig1.update_traces(texttemplate='%{text:.2s}',  textposition=\"outside\",marker_color='teal')\nfig1.update_layout(showlegend=False)\nfig1","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:59:21.461082Z","iopub.execute_input":"2022-01-15T07:59:21.461855Z","iopub.status.idle":"2022-01-15T07:59:22.593568Z","shell.execute_reply.started":"2022-01-15T07:59:21.461817Z","shell.execute_reply":"2022-01-15T07:59:22.592374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Comparison of Non English languages with comments appearing more than 50 times","metadata":{}},{"cell_type":"code","source":"dfq1 = df.query(\"Language != 'English' and Language != 'un'\").query(\"Count >= 50\")\nfig1 =px.scatter(dfq1, y=\"Language\", x=\"Count\", title=\"Count of non-English Language\", size=\"Count\", color=\"Language\", log_x=True, size_max=60)\nfig1.update_traces(mode=\"markers\")\nfig1.update_layout(showlegend=True)\nfig1","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:59:25.047249Z","iopub.execute_input":"2022-01-15T07:59:25.047553Z","iopub.status.idle":"2022-01-15T07:59:25.217463Z","shell.execute_reply.started":"2022-01-15T07:59:25.047523Z","shell.execute_reply":"2022-01-15T07:59:25.216521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Comment Word Distribution","metadata":{}},{"cell_type":"code","source":"import plotly.figure_factory as ff\n\ndef new_len(x):\n    if type(x) is str:\n        return len(x.split())\n    else:\n        return 0\n\ntrain[\"comment_words\"] = train[\"comment_text\"].apply(new_len)\nnums = train.query(\"comment_words != 0 and comment_words < 200\")[\"comment_words\"]\nfig = ff.create_distplot(hist_data=[nums],group_labels=[\"All comments\"],colors=[\"indigo\"])\n\nfig.update_layout(title_text=\"Word distribution per Comment\", xaxis_title=\"Comment words\", showlegend=False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:59:29.037923Z","iopub.execute_input":"2022-01-15T07:59:29.038468Z","iopub.status.idle":"2022-01-15T07:59:39.599375Z","shell.execute_reply.started":"2022-01-15T07:59:29.038433Z","shell.execute_reply":"2022-01-15T07:59:39.598373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Average Comment Words vs Language\nLanguages where the average number of words in comments is less than 200","metadata":{}},{"cell_type":"code","source":"import plotly.graph_objects as go\n\ndfaa = pd.DataFrame(np.transpose([lang_list, train.groupby(\"language\").mean()[\"comment_words\"]]))\ndfaa.columns = [\"Language\", \"avg_comment_words\"]\ndfaa[\"avg_comment_words\"] = dfaa[\"avg_comment_words\"].apply(float)\ndfaa = dfaa.query(\"avg_comment_words < 200\")\nfig = go.Figure()\nfig.add_trace(go.Bar(y=dfaa[\"avg_comment_words\"], x=dfaa[\"Language\"]))\nfig.update_layout(xaxis_title=\"Average Count of words\", yaxis_title=\"Language\", title_text=\"Language Versus Average Number of Words in comments\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T08:02:34.503463Z","iopub.execute_input":"2022-01-15T08:02:34.503890Z","iopub.status.idle":"2022-01-15T08:02:34.610281Z","shell.execute_reply.started":"2022-01-15T08:02:34.503857Z","shell.execute_reply":"2022-01-15T08:02:34.607552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Languages where the average number of words in comments is more than 200","metadata":{}},{"cell_type":"code","source":"import plotly.graph_objects as go\n\ndfab = pd.DataFrame(np.transpose([lang_list, train.groupby(\"language\").mean()[\"comment_words\"]]))\ndfab.columns = [\"Language\", \"avg_comment_words\"]\ndfab[\"avg_comment_words\"] = dfab[\"avg_comment_words\"].apply(float)\ndfab = dfab.query(\"avg_comment_words > 200\")\nfig = go.Figure()\nfig.add_trace(go.Bar(y=dfab[\"avg_comment_words\"], x=dfab[\"Language\"]))\nfig.update_layout(xaxis_title=\"Average Count of words\", yaxis_title=\"Language\", title_text=\"Language Versus Average Number of Words in comments\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T08:02:47.647315Z","iopub.execute_input":"2022-01-15T08:02:47.647629Z","iopub.status.idle":"2022-01-15T08:02:47.747762Z","shell.execute_reply.started":"2022-01-15T08:02:47.647597Z","shell.execute_reply":"2022-01-15T08:02:47.746567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk.sentiment.vader import SentimentIntensityAnalyzer\n\ndef polarity_score(x):\n    if type(x) == str:\n        return SIA.polarity_scores(x)\n    else:\n        return 1000\n    \nSIA = SentimentIntensityAnalyzer()\ntrain[\"polarity\"] = train[\"comment_text\"].progress_apply(polarity_score)","metadata":{"execution":{"iopub.status.busy":"2022-01-15T08:02:55.429301Z","iopub.execute_input":"2022-01-15T08:02:55.429897Z","iopub.status.idle":"2022-01-15T08:09:04.973435Z","shell.execute_reply.started":"2022-01-15T08:02:55.429862Z","shell.execute_reply":"2022-01-15T08:09:04.972537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(go.Histogram(x=[pols[\"neg\"] for pols in train[\"polarity\"] if pols[\"neg\"] != 0], marker=dict(color='teal')))\n\nfig.update_layout(title_text=\"Negative sentiment\", template=\"simple_white\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T08:09:11.559599Z","iopub.execute_input":"2022-01-15T08:09:11.559911Z","iopub.status.idle":"2022-01-15T08:09:14.609701Z","shell.execute_reply.started":"2022-01-15T08:09:11.559883Z","shell.execute_reply":"2022-01-15T08:09:14.608893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = go.Figure(go.Histogram(x=[p[\"pos\"] for p in train[\"polarity\"] if p[\"pos\"] != 0], marker=dict(color='darkblue')))\nfig.update_layout(title_text=\"Positive sentiment\", template=\"simple_white\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T08:09:24.342043Z","iopub.execute_input":"2022-01-15T08:09:24.342326Z","iopub.status.idle":"2022-01-15T08:09:27.389978Z","shell.execute_reply.started":"2022-01-15T08:09:24.342282Z","shell.execute_reply":"2022-01-15T08:09:27.389125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Toxicity in Comparison with Negativity and Positivity","metadata":{}},{"cell_type":"code","source":"#train[\"negativity\"] = train[\"polarity\"].apply(lambda x: x[\"neg\"])\n#one = train.query(\"toxic == 1\")[\"negativity\"]\n#zero = train.query(\"toxic == 0\")[\"negativity\"]\n\n#fig = ff.create_distplot(hist_data=[one, zero],group_labels=[\"Toxic\", \"Non-toxic\"],colors=[\"slategrey\", \"dodgerblue\"], show_hist=False)\n\n#fig.update_layout(title_text=\"Negativity vs. Toxicity\", xaxis_title=\"Negativity\", template=\"simple_white\")\n#fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T07:15:50.551776Z","iopub.execute_input":"2022-01-15T07:15:50.552427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train[\"positivity\"] = train[\"polarity\"].apply(lambda x: x[\"pos\"])\n#nums_1 = train.sample(frac=0.1).query(\"toxic == 1\")[\"positivity\"]\n#nums_2 = train.sample(frac=0.1).query(\"toxic == 0\")[\"positivity\"]\n\n#fig = ff.create_distplot(hist_data=[nums_1, nums_2],group_labels=[\"Toxic\", \"Non-toxic\"], colors=[\"dodgerblue\", \"purple\"], show_hist=False)\n\n#fig.update_layout(title_text=\"Positivity vs. Toxicity\", xaxis_title=\"Positivity\", template=\"simple_white\")\n#fig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Readability","metadata":{}},{"cell_type":"code","source":"pip install textstat","metadata":{"execution":{"iopub.status.busy":"2022-01-15T08:10:13.700604Z","iopub.execute_input":"2022-01-15T08:10:13.700931Z","iopub.status.idle":"2022-01-15T08:10:24.395184Z","shell.execute_reply.started":"2022-01-15T08:10:13.700884Z","shell.execute_reply":"2022-01-15T08:10:24.393933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import textstat\ntrain[\"flesch_reading_ease\"] = train[\"comment_text\"].progress_apply(textstat.flesch_reading_ease)\nfig = go.Figure(go.Histogram(x=train.query(\"flesch_reading_ease > 0\")[\"flesch_reading_ease\"], marker=dict(color='dodgerblue')))\n\nfig.update_layout(title_text=\"Flesch reading ease\", template=\"simple_white\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-15T08:10:24.397831Z","iopub.execute_input":"2022-01-15T08:10:24.398161Z","iopub.status.idle":"2022-01-15T08:12:47.964072Z","shell.execute_reply.started":"2022-01-15T08:10:24.398116Z","shell.execute_reply":"2022-01-15T08:12:47.962573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" ","metadata":{},"execution_count":null,"outputs":[]}]}