{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom wordcloud import WordCloud ,STOPWORDS\nfrom collections import Counter","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"G:/Jigsaw-toxic-comment-classification/train.csv\") \ntest = pd.read_csv(\"G:/Jigsaw-toxic-comment-classification/test.csv\")\ntest_y = pd.read_csv(\"G:/Jigsaw-toxic-comment-classification/test_labels.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('train shape:',train.shape)\ntrain.head(10)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('test shape:',test.shape)\ntest.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_y.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# marking comments without any tags as \"clean\"\ntag_sums = train.iloc[:,2:].sum(axis=1)\ntrain['clean'] = (tag_sums==0)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Check for missing values in Train dataset\")\nnull_train = train.isnull().sum()\nprint(null_train)\nprint(\"Check for missing values in Train dataset\")\nnull_test = test.isnull().sum()\nprint(null_test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example of clean comment\ntrain['comment_text'][0]\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example of toxic comment\ntrain[train.toxic == 1].iloc[1, 1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# example of identity_hate comment\ntrain[train.identity_hate == 1].iloc[1, 1]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# just a random comment\ntrain['comment_text'][157718]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_count = train[train.columns[2:]].sum()\nlabel_count","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,4))\nsns.barplot(x= label_count.index, y = label_count.values, palette= sns.color_palette(\"Reds\"))\nplt.xticks(rotation=90)\nplt.title('Class distribution', fontsize=12)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"comment_len = train.comment_text.str.len()\n\n# plot the distribution of comment lengths\nplt.figure(figsize=(8,4))\nsns.histplot(comment_len, kde=False, bins=50, color=\"red\")\nplt.xlabel(\"Comment Length (Number of words)\", fontsize=12)\nplt.ylabel(\"Number of Comments\", fontsize=12)\nplt.title(\"Distribution of comment Lengths\", fontsize=12)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = ['toxic', 'severe_toxic', 'obscene', 'threat', 'insult', 'identity_hate']\n\nfig, ax = plt.subplots(nrows=3, ncols=2, figsize=(15,10), sharex=True)\naxes =ax.ravel()\n\nfor i in range(6):\n    comments = train.loc[train[labels[i]] == 1, :]\n    comment_len = [len(comment.split()) for comment in comments[\"comment_text\"]]\n    sns.histplot(comment_len, ax=axes[i], bins = 50, color=\"red\")\n    axes[i].title.set_text(labels[i])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clean words\nsubset=train[train.clean==True]\ntext = \" \".join(i for i in subset.comment_text)\nstopwords = set(STOPWORDS)\nwordcloud = WordCloud(stopwords=stopwords, colormap=\"Greens\").generate(text)\nplt.figure( figsize=(8,4))\nplt.imshow(wordcloud, interpolation='bilinear')\nplt.axis(\"off\")\nplt.title(\"Words frequented in Clean Comments\", fontsize=20)\nplt.show()","metadata":{},"execution_count":null,"outputs":[]}]}