{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Team members:\n- Asmaa Mohy\n- Aya Hany\n- Fatma Marzouk\n- Ali Daebis","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T22:53:00.004909Z","iopub.execute_input":"2022-07-11T22:53:00.005364Z","iopub.status.idle":"2022-07-11T22:53:00.035572Z","shell.execute_reply.started":"2022-07-11T22:53:00.005252Z","shell.execute_reply":"2022-07-11T22:53:00.034549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport re\nimport nltk\nfrom nltk.stem.porter import PorterStemmer\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\nimport string\nnltk.download('stopwords')\nstop_words = stopwords.words('english')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:53:00.173880Z","iopub.execute_input":"2022-07-11T22:53:00.175605Z","iopub.status.idle":"2022-07-11T22:53:01.885954Z","shell.execute_reply.started":"2022-07-11T22:53:00.175560Z","shell.execute_reply":"2022-07-11T22:53:01.884978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training data\ntrain = pd.read_csv('../input/jigsaw-toxic-comment-classification-challenge/train.csv.zip')\n# Testing data\ntest = pd.read_csv('../input/jigsaw-toxic-comment-classification-challenge/test.csv.zip')\n# sample \nsample = pd.read_csv('../input/jigsaw-toxic-comment-classification-challenge/sample_submission.csv.zip')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:53:01.888225Z","iopub.execute_input":"2022-07-11T22:53:01.888669Z","iopub.status.idle":"2022-07-11T22:53:05.572590Z","shell.execute_reply.started":"2022-07-11T22:53:01.888628Z","shell.execute_reply":"2022-07-11T22:53:05.571615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train['comment_text'].iloc[125698]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:53:05.573988Z","iopub.execute_input":"2022-07-11T22:53:05.574340Z","iopub.status.idle":"2022-07-11T22:53:05.579160Z","shell.execute_reply.started":"2022-07-11T22:53:05.574289Z","shell.execute_reply":"2022-07-11T22:53:05.578274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_colwidth', None)\ntrain['comment_text'].sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:53:05.580951Z","iopub.execute_input":"2022-07-11T22:53:05.581808Z","iopub.status.idle":"2022-07-11T22:53:05.606095Z","shell.execute_reply.started":"2022-07-11T22:53:05.581770Z","shell.execute_reply":"2022-07-11T22:53:05.605268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# problems in data\n#[User:Cirt]]\n#200.83.101.199\n#(Romania)\n#hi moron\n# URLs","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:14:55.638564Z","iopub.execute_input":"2022-07-11T22:14:55.638944Z","iopub.status.idle":"2022-07-11T22:14:55.643743Z","shell.execute_reply.started":"2022-07-11T22:14:55.638893Z","shell.execute_reply":"2022-07-11T22:14:55.642569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Label column names\nlabels = list(train.columns[2:])\nlabels","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:14:56.470980Z","iopub.execute_input":"2022-07-11T22:14:56.471333Z","iopub.status.idle":"2022-07-11T22:14:56.479538Z","shell.execute_reply.started":"2022-07-11T22:14:56.471306Z","shell.execute_reply":"2022-07-11T22:14:56.478429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label = train[['toxic', 'severe_toxic' , 'obscene' , 'threat' , 'insult' , 'identity_hate']]\nprint(label.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:53:32.598801Z","iopub.execute_input":"2022-07-11T22:53:32.599166Z","iopub.status.idle":"2022-07-11T22:53:32.618790Z","shell.execute_reply.started":"2022-07-11T22:53:32.599137Z","shell.execute_reply":"2022-07-11T22:53:32.617542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n## Let us find out the frequency of occurence of multilabelled data\n\n  - ct1 counts samples having atleast one label\n  - ct2 counts samples having 2 or more than 2 labels\n\n","metadata":{}},{"cell_type":"code","source":"ct1,ct2 = 0,0\nfor i in range(label.shape[0]):\n    ct = np.count_nonzero(label.iloc[i])\n    if ct :\n        ct1 = ct1+1\n    if ct>1 :\n        ct2 = ct2+1\nprint(ct1)\nprint(ct2)\nprint(train.shape[0]-(ct1+ct2))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:01:55.660782Z","iopub.execute_input":"2022-07-11T14:01:55.661189Z","iopub.status.idle":"2022-07-11T14:02:16.780106Z","shell.execute_reply.started":"2022-07-11T14:01:55.661156Z","shell.execute_reply":"2022-07-11T14:02:16.778704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lengths of texts","metadata":{}},{"cell_type":"code","source":"x = [len(train['comment_text'][i]) for i in range(train['comment_text'].shape[0])]\n\nprint('average length of comment: {:.3f}'.format(sum(x)/len(x)) )\nbins = [1,200,400,600,800,1000,1200]\nplt.hist(x, bins=bins)\nplt.xlabel('Length of comments')\nplt.ylabel('Number of comments')       \nplt.axis([0, 1200, 0, 90000])\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T20:15:33.446031Z","iopub.execute_input":"2022-07-10T20:15:33.446720Z","iopub.status.idle":"2022-07-10T20:15:35.395154Z","shell.execute_reply.started":"2022-07-10T20:15:33.446681Z","shell.execute_reply":"2022-07-10T20:15:35.394281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Number of comments classified as toxic,severe_toxic,....etc depending on their lengths","metadata":{}},{"cell_type":"code","source":"label.head(1)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T20:15:35.397527Z","iopub.execute_input":"2022-07-10T20:15:35.398182Z","iopub.status.idle":"2022-07-10T20:15:35.410376Z","shell.execute_reply.started":"2022-07-10T20:15:35.398142Z","shell.execute_reply":"2022-07-10T20:15:35.409467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = np.zeros(label.shape)\nfor ix in range(train['comment_text'].shape[0]):\n    l = len(train['comment_text'][ix])\n    if label['toxic'].iloc[ix] :\n        y[ix][0] = l\n    if label['severe_toxic'].iloc[ix] :\n        y[ix][1] = l\n    if label['obscene'].iloc[ix] :\n        y[ix][2] = l\n    if label['threat'].iloc[ix] :\n        y[ix][3] = l\n    if label['insult'].iloc[ix] :\n        y[ix][4] = l\n    if label['identity_hate'].iloc[ix] :\n        y[ix][5] = l\n\nlabelsplt = ['toxic','severe_toxic','obscene','threat','insult','identity_hate']\ncolor = ['red','green','blue','yellow','orange','chartreuse']  \nplt.figure(figsize=(20,20))\nplt.hist(y,bins = bins,label = labelsplt,color = color)\nplt.axis([0, 1200, 0, 8000])\nplt.xlabel('Length of comments')\nplt.ylabel('Number of comments') \nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T20:15:35.413142Z","iopub.execute_input":"2022-07-10T20:15:35.414123Z","iopub.status.idle":"2022-07-10T20:15:48.231338Z","shell.execute_reply.started":"2022-07-10T20:15:35.414086Z","shell.execute_reply":"2022-07-10T20:15:48.230260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Number of comments classified as toxic,severe_toxic,....etc overall data","metadata":{}},{"cell_type":"code","source":"plt.barh(labelsplt,train[labels].sum(axis = 0),color = color)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T20:15:48.236268Z","iopub.execute_input":"2022-07-10T20:15:48.238789Z","iopub.status.idle":"2022-07-10T20:15:48.484059Z","shell.execute_reply.started":"2022-07-10T20:15:48.238748Z","shell.execute_reply":"2022-07-10T20:15:48.483162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get value counts for each class\ntrain[labels].sum(axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T20:15:48.489141Z","iopub.execute_input":"2022-07-10T20:15:48.491520Z","iopub.status.idle":"2022-07-10T20:15:48.506910Z","shell.execute_reply.started":"2022-07-10T20:15:48.491481Z","shell.execute_reply":"2022-07-10T20:15:48.505811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get number of comments that have been classified (toxic)\n# comments_classified = sum(train[labels].sum(axis = 1) !=0)\n# comments_classified","metadata":{"execution":{"iopub.status.busy":"2022-07-10T20:15:48.511246Z","iopub.execute_input":"2022-07-10T20:15:48.513610Z","iopub.status.idle":"2022-07-10T20:15:48.519659Z","shell.execute_reply.started":"2022-07-10T20:15:48.513571Z","shell.execute_reply":"2022-07-10T20:15:48.518636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# comments_not_classified = comments - comments_classified\n# print(f\"there are {comments_not_classified} not toxic comments in the dataset\")","metadata":{"execution":{"iopub.status.busy":"2022-07-10T20:15:48.521523Z","iopub.execute_input":"2022-07-10T20:15:48.521887Z","iopub.status.idle":"2022-07-10T20:15:48.531384Z","shell.execute_reply.started":"2022-07-10T20:15:48.521847Z","shell.execute_reply":"2022-07-10T20:15:48.530395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # adding not toxic column\n# condition = train[labels].sum(axis = 1) == 0\n# train['not_toxic'] = np.where(condition, 1,0)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T19:33:26.244395Z","iopub.execute_input":"2022-07-10T19:33:26.245101Z","iopub.status.idle":"2022-07-10T19:33:26.252808Z","shell.execute_reply.started":"2022-07-10T19:33:26.245061Z","shell.execute_reply":"2022-07-10T19:33:26.251913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # checking if there is an undefined toxicity type\n# sum(train[labels].sum(axis = 1) == 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T19:33:26.257502Z","iopub.execute_input":"2022-07-10T19:33:26.257766Z","iopub.status.idle":"2022-07-10T19:33:26.263801Z","shell.execute_reply.started":"2022-07-10T19:33:26.257742Z","shell.execute_reply":"2022-07-10T19:33:26.262830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # adding toxicity undefined column\n# condition = train[labels].sum(axis = 1) == 1\n# train['undefined_toxic'] = np.where(condition, 1,0) ","metadata":{"execution":{"iopub.status.busy":"2022-07-10T19:33:26.264924Z","iopub.execute_input":"2022-07-10T19:33:26.265190Z","iopub.status.idle":"2022-07-10T19:33:26.275685Z","shell.execute_reply.started":"2022-07-10T19:33:26.265166Z","shell.execute_reply":"2022-07-10T19:33:26.274607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # checking if there is a comment classified with a toxic type without classifying it as toxic\n# train['toxicity_type_defined'] = train[['insult','obscene','identity_hate','threat']].max(axis=1)\n# condition = (train['toxicity_type_defined']==1) & (train['toxic']==0)\n# train['soft_toxic'] = np.where(condition, 1,0)\n# train.drop(['toxicity_type_defined'], axis = 1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T19:33:26.277397Z","iopub.execute_input":"2022-07-10T19:33:26.277810Z","iopub.status.idle":"2022-07-10T19:33:26.288747Z","shell.execute_reply.started":"2022-07-10T19:33:26.277776Z","shell.execute_reply":"2022-07-10T19:33:26.287833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # adding new labels\n# labels.extend(['not_toxic', 'undefined_toxic', 'soft_toxic'])","metadata":{"execution":{"iopub.status.busy":"2022-07-10T19:33:26.290284Z","iopub.execute_input":"2022-07-10T19:33:26.290674Z","iopub.status.idle":"2022-07-10T19:33:26.298557Z","shell.execute_reply.started":"2022-07-10T19:33:26.290638Z","shell.execute_reply":"2022-07-10T19:33:26.297517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label_counts = train[labels].sum()\n# plt.figure(figsize=(20, 10))\n# ax = sns.barplot(x=label_counts.index, y=label_counts.values)\n# ax.set_yscale(\"log\")\n# ax.tick_params(labelsize=15)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T19:33:26.300080Z","iopub.execute_input":"2022-07-10T19:33:26.300416Z","iopub.status.idle":"2022-07-10T19:33:26.309240Z","shell.execute_reply.started":"2022-07-10T19:33:26.300384Z","shell.execute_reply":"2022-07-10T19:33:26.308242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# heatmap_data = train[labels]\n# plt.figure(figsize=(10, 10))\n# ax = sns.heatmap(heatmap_data.corr(), cmap='coolwarm', annot=True)\n# ax.tick_params(labelsize=10)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T19:33:26.311048Z","iopub.execute_input":"2022-07-10T19:33:26.311374Z","iopub.status.idle":"2022-07-10T19:33:26.318915Z","shell.execute_reply.started":"2022-07-10T19:33:26.311347Z","shell.execute_reply":"2022-07-10T19:33:26.317969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Confirm that all severly toxic comments (n=1595) are toxic:\n# train.loc[train['severe_toxic']==1,'toxic'].sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T19:33:26.320485Z","iopub.execute_input":"2022-07-10T19:33:26.320835Z","iopub.status.idle":"2022-07-10T19:33:26.328919Z","shell.execute_reply.started":"2022-07-10T19:33:26.320800Z","shell.execute_reply":"2022-07-10T19:33:26.328013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def to_lowercase(text):\n    return text.lower()\n\n# Remove website links\ndef remove_links(text):\n    template = re.compile(r'https?://\\S+|www\\.\\S+') \n    text = template.sub(r'', text)\n    return text\n\n# Remove HTML tags\ndef remove_html(text):\n    template = re.compile(r'<[^>]*>') \n    text = template.sub(r'', text)\n    return text\n\ndef text2words(text):\n      return word_tokenize(text)\n\n# Remove stopwords\ndef remove_stopwords(words, stop_words):\n    return [word for word in words if word not in stop_words]\n\n# Remove none ascii characters\ndef remove_non_ascii(text):\n    template = re.compile(r'[^\\x00-\\x7E]+') \n    text = template.sub(r'', text)\n    return text\n\n# Replace none printable characters\ndef remove_non_printable(text):\n    template = re.compile(r'[\\x00-\\x0F]+') \n    text = template.sub(r' ', text)\n    return text\n\n# Remove special characters\ndef remove_special_chars(text):\n        text = re.sub(\"'s\", '', text)\n        template = re.compile('[\"#$%&\\'()\\*\\+-/:;<=>@\\[\\]\\\\\\\\^_`{|}~]') \n        text = template.sub(r' ', text)\n        return text\n\n# Replace multiple punctuation \ndef replace_multiplt_punc(text):\n        text = re.sub('[.!?]{2,}', '.', text)\n        text = re.sub(',+', ',', text) \n        return text\n\n\ndef remove_punctuation(text):\n    \"\"\"Remove punctuation from list of tokenized words\"\"\"\n    translator = str.maketrans('', '', string.punctuation)\n    return text.translate(translator)\n\n\n# Remove numbers\ndef remove_numbers(text):\n        text = re.sub('\\d+', ' ', text)\n        return text\n\ndef handle_spaces(text):\n    # Remove extra spaces\n    text = re.sub('\\s+', ' ', text)\n    \n    # Remove spaces at the beginning and at the end of string\n    text = text.strip() \n    \n    return text\n\ndef stem_words(words):\n    \"\"\"Stem words in text\"\"\"\n    stemmer = PorterStemmer()\n    return [stemmer.stem(word) for word in words]\n\ndef lemmatize_words(words):\n    \"\"\"Lemmatize words in text\"\"\"\n\n    lemmatizer = WordNetLemmatizer()\n    return [lemmatizer.lemmatize(word) for word in words]\n\ndef lemmatize_verbs(words):\n    \"\"\"Lemmatize verbs in text\"\"\"\n\n    lemmatizer = WordNetLemmatizer()\n    return ([lemmatizer.lemmatize(word, pos='v') for word in words])\n\ndef remove_pattern(text): \n    # remove hi moron \n    text= re.sub(r'(hi)(.*)\\1', r'\\1', text)\n    # remove duplicate words\n    text= re.sub(r\"\\b(\\w+)(?:\\W+\\1\\b)+\",r'\\1', text,flags=re.IGNORECASE)\n    # remove [User:Cirt]] \n    text= re.sub(r\"\\[.*?\\]\", ' ', text)\n    # remove \\n\\n\n    text= re.sub(r\"\\n\", ' ', text)\n    return text\n\ndef clean_text( text):\n    text = remove_pattern(text)\n    text = remove_links(text)\n    text = remove_html(text)\n    text = remove_special_chars(text)\n    text = remove_non_ascii(text)\n    text = remove_non_printable(text)\n    text = remove_numbers(text)\n    text = remove_punctuation(text)\n    text = to_lowercase(text)\n    text = handle_spaces(text)\n    words = text2words(text)\n    words = remove_stopwords(words, stop_words)\n    #words = stem_words(words) #either stem or lemmatize\n    words = lemmatize_words(words)\n    words = lemmatize_verbs(words)\n\n    return ' '.join(words)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:53:56.506455Z","iopub.execute_input":"2022-07-11T22:53:56.506825Z","iopub.status.idle":"2022-07-11T22:53:56.528361Z","shell.execute_reply.started":"2022-07-11T22:53:56.506793Z","shell.execute_reply":"2022-07-11T22:53:56.527324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train['comment_text'].iloc[2])\nprint('-------------------')\nsample = clean_text(train['comment_text'].iloc[2])\nsample","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:53:57.952244Z","iopub.execute_input":"2022-07-11T22:53:57.954635Z","iopub.status.idle":"2022-07-11T22:53:59.819126Z","shell.execute_reply.started":"2022-07-11T22:53:57.954586Z","shell.execute_reply":"2022-07-11T22:53:59.818240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def return_tweets(df):\n    texts=[]\n    for index, item in df.drop(df.columns.difference(['comment_text']), axis=1).iterrows():\n        message = item[\"comment_text\"]\n        texts.append(str(message))\n    return texts","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:54:02.391175Z","iopub.execute_input":"2022-07-11T22:54:02.391843Z","iopub.status.idle":"2022-07-11T22:54:02.398532Z","shell.execute_reply.started":"2022-07-11T22:54:02.391804Z","shell.execute_reply":"2022-07-11T22:54:02.397371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_texts = return_tweets(train)\ntest_texts = return_tweets(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:54:02.676176Z","iopub.execute_input":"2022-07-11T22:54:02.676896Z","iopub.status.idle":"2022-07-11T22:54:16.016739Z","shell.execute_reply.started":"2022-07-11T22:54:02.676856Z","shell.execute_reply":"2022-07-11T22:54:16.015683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_corpus(corpus):\n    return [clean_text(t) for t in corpus]\ntrain_texts = clean_corpus(train_texts)\ntest_texts = clean_corpus(test_texts)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:54:16.018677Z","iopub.execute_input":"2022-07-11T22:54:16.019020Z","iopub.status.idle":"2022-07-11T22:58:52.642664Z","shell.execute_reply.started":"2022-07-11T22:54:16.018985Z","shell.execute_reply":"2022-07-11T22:58:52.641667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_texts[:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:58:52.644064Z","iopub.execute_input":"2022-07-11T22:58:52.644444Z","iopub.status.idle":"2022-07-11T22:58:52.651910Z","shell.execute_reply.started":"2022-07-11T22:58:52.644408Z","shell.execute_reply":"2022-07-11T22:58:52.650888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = [len(train_texts[i]) for i in range(len(train_texts))]\n\nprint('average length of comment: {:.3f}'.format(sum(x)/len(x)) )\nbins = [1,200,400,600,800,1000,1200]\nplt.hist(x, bins=bins)\nplt.xlabel('Length of comments')\nplt.ylabel('Number of comments')       \nplt.axis([0, 1200, 0, 90000])\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:58:52.654427Z","iopub.execute_input":"2022-07-11T22:58:52.655035Z","iopub.status.idle":"2022-07-11T22:58:53.577189Z","shell.execute_reply.started":"2022-07-11T22:58:52.654999Z","shell.execute_reply":"2022-07-11T22:58:53.576238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"maxlen=600","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:58:53.578690Z","iopub.execute_input":"2022-07-11T22:58:53.579040Z","iopub.status.idle":"2022-07-11T22:58:53.583346Z","shell.execute_reply.started":"2022-07-11T22:58:53.579006Z","shell.execute_reply":"2022-07-11T22:58:53.582363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## BoW ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.text import Tokenizer  \ntok = Tokenizer(num_words=1000, oov_token='UNK')\n#tok = Tokenizer(oov_token='UNK')\ntok.fit_on_texts(train_texts + test_texts)\n# Extract binary BoW features\n# x_train = tok.texts_to_matrix(train_texts, mode='tfidf')\n# x_test = tok.texts_to_matrix(test_texts, mode='tfidf')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:59:10.762687Z","iopub.execute_input":"2022-07-11T22:59:10.763035Z","iopub.status.idle":"2022-07-11T22:59:25.231745Z","shell.execute_reply.started":"2022-07-11T22:59:10.763006Z","shell.execute_reply":"2022-07-11T22:59:25.230696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = np.asarray(label.values).astype('float32')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:59:25.233743Z","iopub.execute_input":"2022-07-11T22:59:25.234453Z","iopub.status.idle":"2022-07-11T22:59:25.240104Z","shell.execute_reply.started":"2022-07-11T22:59:25.234416Z","shell.execute_reply":"2022-07-11T22:59:25.239173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(x_train.shape)\n# print(y_train.shape)\n# print(x_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:23:16.474889Z","iopub.execute_input":"2022-07-11T14:23:16.475512Z","iopub.status.idle":"2022-07-11T14:23:16.488841Z","shell.execute_reply.started":"2022-07-11T14:23:16.475468Z","shell.execute_reply":"2022-07-11T14:23:16.487384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### vocab_inp_size = len(tok.word_index)+1","metadata":{}},{"cell_type":"code","source":"vocab_inp_size = len(tok.word_index) + 1\n#vocab_inp_size = 1000","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:02:08.505858Z","iopub.execute_input":"2022-07-11T20:02:08.506431Z","iopub.status.idle":"2022-07-11T20:02:08.510704Z","shell.execute_reply.started":"2022-07-11T20:02:08.506399Z","shell.execute_reply":"2022-07-11T20:02:08.509692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_inp_size","metadata":{"execution":{"iopub.status.busy":"2022-07-11T20:02:10.661361Z","iopub.execute_input":"2022-07-11T20:02:10.662418Z","iopub.status.idle":"2022-07-11T20:02:10.668748Z","shell.execute_reply.started":"2022-07-11T20:02:10.662372Z","shell.execute_reply":"2022-07-11T20:02:10.667453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Split train data","metadata":{}},{"cell_type":"code","source":"# from sklearn.model_selection import train_test_split\n# tr_X, val_X, tr_y, val_y = train_test_split(x_train, y_train, train_size=0.90, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:23:16.518076Z","iopub.execute_input":"2022-07-11T14:23:16.518923Z","iopub.status.idle":"2022-07-11T14:23:17.077450Z","shell.execute_reply.started":"2022-07-11T14:23:16.518879Z","shell.execute_reply":"2022-07-11T14:23:17.076168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(tr_X.shape)\n# print(tr_y.shape)\n# print(val_X.shape)\n# print(val_y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:23:17.079216Z","iopub.execute_input":"2022-07-11T14:23:17.079745Z","iopub.status.idle":"2022-07-11T14:23:17.088650Z","shell.execute_reply.started":"2022-07-11T14:23:17.079665Z","shell.execute_reply":"2022-07-11T14:23:17.086956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build baseline Model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import models\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import losses\nfrom tensorflow.keras import metrics\nfrom tensorflow.keras import optimizers","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:59:25.241456Z","iopub.execute_input":"2022-07-11T22:59:25.242483Z","iopub.status.idle":"2022-07-11T22:59:25.250093Z","shell.execute_reply.started":"2022-07-11T22:59:25.242447Z","shell.execute_reply":"2022-07-11T22:59:25.249228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = models.Sequential()\n# model.add(layers.Dense(16, activation='relu', input_shape=(1000,)))\n# model.add(layers.Dense(8, activation='relu'))\n# model.add(layers.Dense(6, activation='sigmoid'))\n\n# model.compile(optimizer=optimizers.Adam(),\n#               loss=losses.binary_crossentropy,\n#               metrics=['AUC'])\n\n# history = model.fit(tr_X,\n#                     tr_y,\n#                     epochs=20,\n#                     batch_size=128,\n#                     validation_data=(val_X, val_y))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:59:25.252317Z","iopub.execute_input":"2022-07-11T22:59:25.252760Z","iopub.status.idle":"2022-07-11T22:59:25.260432Z","shell.execute_reply.started":"2022-07-11T22:59:25.252724Z","shell.execute_reply":"2022-07-11T22:59:25.259509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# acc = history.history['auc']\n# val_acc = history.history['val_auc']\n# loss = history.history['loss']\n# val_loss = history.history['val_loss']\n\n# epochs = range(1, len(acc) + 1)\n\n# # \"bo\" is for \"blue dot\"\n# plt.plot(epochs, loss, 'bo', label='Training loss')\n# # b is for \"solid blue line\"\n# plt.plot(epochs, val_loss, 'b', label='Validation loss')\n# plt.title('Training and validation loss')\n# plt.xlabel('Epochs')\n# plt.ylabel('Loss')\n# plt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T22:59:25.261875Z","iopub.execute_input":"2022-07-11T22:59:25.262929Z","iopub.status.idle":"2022-07-11T22:59:25.271668Z","shell.execute_reply.started":"2022-07-11T22:59:25.262892Z","shell.execute_reply":"2022-07-11T22:59:25.270734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.plot(epochs, acc, 'bo', label='Training auc')\n# plt.plot(epochs, val_acc, 'b', label='Validation auc')\n# plt.title('Training and validation accuracy')\n# plt.xlabel('Epochs')\n# plt.ylabel('roc_auc')\n# plt.legend()\n\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:25:41.210807Z","iopub.execute_input":"2022-07-11T14:25:41.211621Z","iopub.status.idle":"2022-07-11T14:25:41.459943Z","shell.execute_reply.started":"2022-07-11T14:25:41.211573Z","shell.execute_reply":"2022-07-11T14:25:41.458622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = models.Sequential()\n# model.add(layers.Dense(16, activation='relu', input_shape=(1000,)))\n# model.add(layers.Dense(8, activation='relu'))\n# model.add(layers.Dense(6, activation='sigmoid'))\n\n# model.compile(optimizer=optimizers.Adam(),\n#               loss=losses.binary_crossentropy,\n#               metrics=['AUC'])\n\n# history = model.fit(tr_X,\n#                     tr_y,\n#                     epochs=5,\n#                     batch_size=512,\n#                     validation_data=(val_X, val_y))\n# # compute roc_auc score  \n# from sklearn.metrics import roc_auc_score\n# y_pred = model.predict(val_X)\n# roc_auc_score(val_y, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T14:25:41.463447Z","iopub.execute_input":"2022-07-11T14:25:41.464546Z","iopub.status.idle":"2022-07-11T14:25:52.573388Z","shell.execute_reply.started":"2022-07-11T14:25:41.464498Z","shell.execute_reply":"2022-07-11T14:25:52.572227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #How to create a submission csv file for Kaggle\n# df_sample = pd.read_csv(\"/kaggle/input/jigsaw-toxic-comment-classification-challenge/sample_submission.csv.zip\")\n# y_pred = model.predict(x_test)\n# df_sample[[\"toxic\", \"severe_toxic\", \"obscene\", \"threat\", \"insult\", \"identity_hate\"]] = y_pred\n# df_sample.to_csv('submission.csv', index=False)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:14:51.201408Z","iopub.status.idle":"2022-07-10T23:14:51.203488Z","shell.execute_reply.started":"2022-07-10T23:14:51.203147Z","shell.execute_reply":"2022-07-10T23:14:51.203170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LSTM","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom sklearn.model_selection import train_test_split\n\nx_train = tok.texts_to_sequences(train_texts)\nx_test = tok.texts_to_sequences(test_texts)\n\ntraining_padded = pad_sequences(x_train,\n                                maxlen=25, \n                                truncating='post', \n                                padding='post'\n                               )\n# #tst_texts\ntest_padded = pad_sequences(x_test,\n                            maxlen=25, \n                            truncating='post', \n                            padding='post'\n                               )\n\n\ntr_X, val_X, tr_y, val_y = train_test_split(training_padded, y_train, train_size=0.90, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T23:27:07.226199Z","iopub.execute_input":"2022-07-11T23:27:07.226917Z","iopub.status.idle":"2022-07-11T23:28:12.379470Z","shell.execute_reply.started":"2022-07-11T23:27:07.226870Z","shell.execute_reply":"2022-07-11T23:28:12.378469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tr_X.shape)\nprint(tr_y.shape)\nprint(val_X.shape)\nprint(val_y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T23:28:12.381469Z","iopub.execute_input":"2022-07-11T23:28:12.381817Z","iopub.status.idle":"2022-07-11T23:28:12.387747Z","shell.execute_reply.started":"2022-07-11T23:28:12.381785Z","shell.execute_reply":"2022-07-11T23:28:12.386591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#embedding_dim = 128\n#max_length = 40\nmodel_lstm = models.Sequential()\nmodel_lstm.add(layers.Embedding(1000,128, input_length=25))\nmodel_lstm.add(layers.LSTM(64))\nmodel_lstm.add(layers.Dense(6, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T23:28:12.389611Z","iopub.execute_input":"2022-07-11T23:28:12.390127Z","iopub.status.idle":"2022-07-11T23:28:12.604927Z","shell.execute_reply.started":"2022-07-11T23:28:12.390087Z","shell.execute_reply":"2022-07-11T23:28:12.603989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_lstm.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T23:28:12.607263Z","iopub.execute_input":"2022-07-11T23:28:12.607655Z","iopub.status.idle":"2022-07-11T23:28:12.613941Z","shell.execute_reply.started":"2022-07-11T23:28:12.607620Z","shell.execute_reply":"2022-07-11T23:28:12.612667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_lstm.compile(loss='binary_crossentropy',\n              optimizer='adam',\n              metrics=['AUC'])\n\nhistory = model_lstm.fit(tr_X,\n                    tr_y,\n                    epochs=5,\n                    validation_data=(val_X, val_y))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T23:28:12.615663Z","iopub.execute_input":"2022-07-11T23:28:12.616027Z","iopub.status.idle":"2022-07-11T23:30:19.648693Z","shell.execute_reply.started":"2022-07-11T23:28:12.615975Z","shell.execute_reply":"2022-07-11T23:30:19.647745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['auc']\nval_acc = history.history['val_auc']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1, len(acc) + 1)\n\n# \"bo\" is for \"blue dot\"\nplt.plot(epochs, loss, 'bo', label='Training loss')\n# b is for \"solid blue line\"\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T23:32:26.534766Z","iopub.execute_input":"2022-07-11T23:32:26.535126Z","iopub.status.idle":"2022-07-11T23:32:26.734058Z","shell.execute_reply.started":"2022-07-11T23:32:26.535096Z","shell.execute_reply":"2022-07-11T23:32:26.733112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(epochs, acc, 'bo', label='Training auc')\nplt.plot(epochs, val_acc, 'b', label='Validation auc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('roc_auc')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T23:32:27.851078Z","iopub.execute_input":"2022-07-11T23:32:27.851693Z","iopub.status.idle":"2022-07-11T23:32:28.058181Z","shell.execute_reply.started":"2022-07-11T23:32:27.851656Z","shell.execute_reply":"2022-07-11T23:32:28.057289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#How to create a submission csv file for Kaggle\ndf_sample = pd.read_csv(\"/kaggle/input/jigsaw-toxic-comment-classification-challenge/sample_submission.csv.zip\")\ny_pred = model_lstm.predict(test_padded)\ndf_sample[[\"toxic\", \"severe_toxic\", \"obscene\", \"threat\", \"insult\", \"identity_hate\"]] = y_pred\ndf_sample.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T23:33:06.066165Z","iopub.execute_input":"2022-07-11T23:33:06.066857Z","iopub.status.idle":"2022-07-11T23:33:14.761318Z","shell.execute_reply.started":"2022-07-11T23:33:06.066822Z","shell.execute_reply":"2022-07-11T23:33:14.760326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}