{"cells":[{"metadata":{},"cell_type":"markdown","source":"## Toxic Comment Classification - Keras Embedding Neural Network ##"},{"metadata":{},"cell_type":"markdown","source":"![](https://encrypted-tbn0.gstatic.com/images?q=tbn:ANd9GcRdm5TStpIRDJEJ0DgT9cR9rYregB9WNHcCI_-dfNvt1Sy4l6DB)"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nprint(os.listdir(\"../input/jigsaw-unintended-bias-in-toxicity-classification\"))\nprint(os.listdir(\"../input/glove840\"))\nprint(os.listdir(\"../input/wikinews\"))\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom numpy import array\nfrom numpy import asarray\nfrom numpy import zeros\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.utils import to_categorical\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.models import Sequential\nfrom keras.layers import Embedding\nfrom keras.layers import Input\nfrom keras.layers import Conv1D\nfrom keras.layers import MaxPooling1D\nfrom keras.layers import Flatten\nfrom keras.layers import Dropout\nfrom keras.layers import Dense\nfrom keras.optimizers import RMSprop\nfrom keras.models import Model\nfrom keras.models import load_model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = pd.read_csv(\"../input/jigsaw-unintended-bias-in-toxicity-classification/train.csv\")\ndf_test = pd.read_csv(\"../input/jigsaw-unintended-bias-in-toxicity-classification/test.csv\")\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of samples in training data\", len(df_train))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create dataframe filtering few particular groups only\nidentity = ['male','female','homosexual_gay_or_lesbian','christian','jewish','muslim','black','white','psychiatric_or_mental_illness']\n\ndef funnewdf():\n    newdf= pd.DataFrame()\n    identity = ['male','female','homosexual_gay_or_lesbian','christian','jewish','muslim','black','white','psychiatric_or_mental_illness']\n    for col in identity:\n        newdf = pd.concat([newdf, df_train[pd.notnull(df_train[col])]], ignore_index=True)\n    return newdf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = funnewdf()\ndf_train=df_train[['id','target','comment_text','male','female','homosexual_gay_or_lesbian','christian','jewish','muslim','black','white','psychiatric_or_mental_illness']]\ndf_train = df_train.drop_duplicates().reset_index()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of samples in training data after filtering -\", len(df_train))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train['comment_text'] = df_train['comment_text'].astype(str) \ndf_test['comment_text'] = df_test['comment_text'].astype(str) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Function to convert toxic value of a column to boolean , 1 being most toxic\ndef convert_to_bool(df, col_name):\n    df[col_name] = np.where(df[col_name] >= 0.5, 1, 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Function to convert into boolean value for whole dataframe\ndef convert_dataframe_to_bool(df):\n    bool_df = df.copy()\n    convert_to_bool(bool_df, 'target')\n    return bool_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = convert_dataframe_to_bool(df_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Initiating Keras tokenizer\ntokenizer = Tokenizer()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Fitting tokenizer on train data and converting tokens to sequence\ntokenizer.fit_on_texts(df_train['comment_text'])\nts_train=tokenizer.texts_to_sequences(df_train['comment_text'])\nX_train_vectorized=pad_sequences(ts_train,maxlen=800,padding='post')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"vocab_size = len(tokenizer.word_index) + 1\nprint(\"Size of corpus in terms of number of tokens to be trained\" ,len(tokenizer.word_index))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Using pretrained Glove embedding vector \nembeddings_index = {}\nf = open('../input/glove840/glove.840B.300d.txt', encoding='utf8')\nfor line in f:\n    values = line.split()\n    word = ''.join(values[:-300])\n    coefs = np.asarray(values[-300:], dtype='float32')\n    embeddings_index[word] = coefs\nf.close()\nprint('Loaded %s word vectors.' % len(embeddings_index))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embedding_matrix = zeros((vocab_size, 300))\nfor word, i in tokenizer.word_index.items():\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"embedding_matrix.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Applying tokenizer on test data\ntokenizer.fit_on_texts(df_test['comment_text'])\nts_test=tokenizer.texts_to_sequences(df_test['comment_text'])\nX_test_vectorized=pad_sequences(ts_test,maxlen=800,padding='post')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_train = df_train['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del df_train , df_test , ts_train , ts_test , embeddings_index\nimport gc\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = Sequential()\ne = Embedding(vocab_size, 300, weights=[embedding_matrix], input_length=800, trainable=False)\nmodel.add(e)\nmodel.add(Conv1D(256, 2, activation='relu', padding='same'))\nmodel.add(MaxPooling1D(5, padding='same'))\nmodel.add(Conv1D(256, 3, activation='relu', padding='same'))\nmodel.add(MaxPooling1D(5, padding='same'))\nmodel.add(Conv1D(256, 4, activation='relu', padding='same'))\nmodel.add(MaxPooling1D(40, padding='same'))\nmodel.add(Flatten())\nmodel.add(Dropout(0.2))\nmodel.add(Dense(256, activation='relu'))\nmodel.add(Dense(2, activation='softmax'))\n#model1.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\nmodel.compile(loss='sparse_categorical_crossentropy',optimizer=RMSprop(lr=0.0005),metrics=['sparse_categorical_accuracy'])\n#model.compile(loss='binary_crossentropy',optimizer=RMSprop(lr=0.0005),metrics=['acc'])\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Model Performance ##"},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(X_train_vectorized, y_train, epochs=3, batch_size=1024, validation_split=0.2, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Prediction on test data for competition submission\npredictions = model.predict(X_test_vectorized)[:,1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = pd.Series(predictions,name=\"prediction\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_id = pd.read_csv(\"../input/jigsaw-unintended-bias-in-toxicity-classification/test.csv\",usecols=['id'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_submission = pd.concat([df_id, preds], axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_submission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_submission.to_csv(\"submission.csv\", columns = df_submission.columns, index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}