{"cells":[{"metadata":{"_uuid":"051d70d956493feee0c6d64651c6a088724dca2a","_execution_state":"idle","trusted":true},"cell_type":"code","source":"## Importing packages\n\n# This R environment comes with all of CRAN and many other helpful packages preinstalled.\n# You can see which packages are installed by checking out the kaggle/rstats docker image: \n# https://github.com/kaggle/docker-rstats\n\nlibrary(tidyverse) # metapackage with lots of helpful functions\n\n## Running code\n\n# In a notebook, you can run a single code cell by clicking in the cell and then hitting \n# the blue arrow to the left, or by clicking in the cell and pressing Shift+Enter. In a script, \n# you can run code by highlighting the code you want to run and then clicking the blue arrow\n# at the bottom of this window.\n\n## Reading in files\n\n# You can access files from datasets you've added to this kernel in the \"../input/\" directory.\n# You can see the files added to this kernel by running the code below. \n\nlist.files(path = \"../input\")\nlibrary(data.table)\ntrain <- fread(\"../input/quora-insincere-questions-classification/train.csv\",nrows=10000)\ntest <- fread(\"../input/quora-insincere-questions-classification//test.csv\")\n\n\n\n\n\n\n# CREATE VALIDATION SET!!!\n\n\n\n\n\n\n\n\n\n\n# Do this later when applying stuff to our test set as well. Still doing EDA at the moment. \n# This to get the data stacked so that we can apply transformations to our train and test. Then we can separate them out using the is_this_test column.\nyTrn <- train[[\"target\"]]\nrow.names(yTrn) <- train[[\"pid\"]]\n#train[, is_this_test := 0]\n#test[, is_this_test := 1]\n#all_data <- rbind(train,test)\n\n## Saving data\n\n\n# If you save any files or images, these will be put in the \"output\" directory. You \n# can see the output directory by committing and running your kernel (using the \n# Commit & Run button) and then checking out the compiled version of your kernel.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"install.packages(\"stringr\")\nlibrary(\"stringr\")\n# str_extract tries to match the text given. If it does then it has the same value. Otherwise it will return empty. Code below returns 1 if there is one, 0 if there is not.\ntemp <- train[,temp := ifelse(is.na(str_extract(question_text, \"&\")),0,1)]\ntemp <- NULL # This just removes a column.\ntable(temp[[\"temp\"]])\n\n# Using this to filter text. This will be useful for Dan M. Just put the code you want to search in the string. You can also apply regex(<regex expression>) if you want to do a regex search. \ntrain[ifelse(is.na(str_extract(question_text, \"&\")),FALSE,TRUE),]\n\n\n\n\n#[,`:=` (prop = sum(target==1)/.N,total = .N),by=ALWAYS][,c(\"ALWAYS\",\"target\",\"prop\",\"total\")])\n# Still fixing this... \n#temp %>% group_by(c(\"ALWAYS\",\"target\")) %>% mutate()\nprint(temp)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"library(\"tidyverse\")\ntemp <- train %>% mutate(text_is_there = ifelse((is.na(str_extract(question_text,\"#\"))) | (str_extract(question_text,\"#\")==\"\"),FALSE,TRUE)) %>% group_by(text_is_there, target) %>% count(1) %>% group_by(text_is_there) %>% mutate(prop = n/sum(n))\n\nprint(temp)\n# This is proportion of target being what it is given that the state of text_is_there.\n# E.g. Prop for FALSE and target = 0 is False & Target == 0 / (Sum of Both Cases with Variable == False)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"install.packages(c(\"tokenizers\",\"textclean\",\"stopwords\"))\nlibrary(\"stopwords\")\nlibrary(\"tokenizers\")\nlibrary(\"textclean\") # Really useful package for this\nprint(head(train))\n\n# Dan M - This is data table syntax to update a row. The comma at the start moves us from editting the row, to editting the column. (row,column). \n# edited_text is the variable which gets created through the update notation \":=\" by applying the replace_contraction function onto question_text.\n# You can just apply the \"=\" function if you just want to create the row temporarily in that line of code but \":=\" declares it permanently.\n# You can see from the summary below that question_text is our column containing all of our questions, so this is just creating the new column edited_text. \n# Subsequent calls on it are just updating the values again using the respective functions but I could just chain them E.g. replace_date(replace_contraction(edited_text), replacement=\"THISISADATE\")\n\n\n# Remove contractions\ntrain[, edited_text := replace_contraction(question_text)]  \n\n# Replace all dates with \"THISISADATE\"\ntrain[, edited_text := replace_date(edited_text, replacement=\"THISISADATE\")]\n\n\n\n# This just removes all punctuation. This is a regex replace text function, (regex code for what to be replaced, replace with, object to apply this to (in this case, column edited_text))\ntrain[, edited_text := gsub(\"[[:punct:]]+\",\"\",edited_text)]\n\n# Convert everything to lowercase for now. Might want to do something more clever with this at a later point. Might be worth us trying to identify pronouns with this.\ntrain[, edited_text := tolower(edited_text)]\n\n# Have not done this but this is a replace Names function as well.\n\n# Again need to filter this more but might be a start\ntrain[, edited_text := replace_curly_quote(edited_text)]\ntrain[, edited_text := replace_non_ascii(edited_text,replacement=\"\",remove.nonconverted=TRUE)]\n\n# Removing all numbers, again would like to filter more but I think it will be hard to get context correct with something like \"one hundred and one\" in such a short time frame.\ntrain[, edited_text := replace_number(edited_text, remove=TRUE)]\ntrain[, edited_text := replace_ordinal(edited_text)] # \"1st\" becomes \"first\"\n\n\n# This replaces $ with \"dollar\", % with \"percent\", \"@\" with \"at\", \"&\" with \"and\", \"w/\" with \"with\". Pound in this refers to the American Version which we call \"hash\" - \"#\"\ntrain[, edited_text := replace_symbol(edited_text, dollar=TRUE, percent=TRUE, at=TRUE, and=TRUE, with=TRUE,pound=FALSE)]\n\n# Need to look for hash tags to see whether they need replacing.\n\n\n# This is assuming time is captured in the regex \"(2[0-3]|[01]?[0-9]):([0-5][0-9])[.:]?([0-5]?[0-9])?\" which looks reasonable to me.\ntrain[, edited_text := replace_time(edited_text, replacement=\"NULL\")]\n\n# Removes URLs\ntrain[, edited_text := replace_url(edited_text)]\n\n# Remove consecutive spaces\ntrain[, edited_text := replace_white(edited_text)]\n\n# Could replace word elongation e.g. \"Heyyyyyyy\" but I think that would be predictive\n# Can use the \"strip\" function in text clean if you need to remove a specific symbol that I have missed\n\n\n# I need to still work these out.\n#wa <- which_are()\n#print(head(wa[[\"digit\"]][data[[\"edited_text\"]]]))\n\ntrain[,tokens := tokenize_word_stems(edited_text, stopwords = stopwords::stopwords(\"en\"))]\ntrain[,all_text := sapply(tokens, function(x) paste(x,collapse=\" \"))]\n#train[,exposure := lapply(tokens, function(x) length(x)[1])]\n                          \n                          \n#print(lapply(head(train[[\"tokens\"]]), function(x) print(x)))\nprint(head(train[[\"all_text\"]]))\n#print(head(train[[\"exposure\"]]))\n#train[[\"tokens\"]]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[, exposure:=sapply(train[[\"tokens\"]], function(x) length(x))]\nprint(train[[\"exposure\"]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"install.packages(\"keras\")\nlibrary(\"keras\")\ntokenizer <- text_tokenizer(num_words=1000000)\ntokenizer %>% fit_text_tokenizer(train[[\"all_text\"]]) # Insert data here.\n\n#print(head(train[[\"tokens\"]]))\nprint(tokenizer$document_count)\n#print(tokenizer$word_index)\ntokenizer$word_index %>% head()\n\ntemp3 <- train\ntemp4 <- yTrn\n#print(temp4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Need to adjust output_dim and max_len?\n#install.packages(\"keras\")\nlibrary(\"keras\")\nmodel <- keras_model_sequential()\n\ntrain <- temp3\nyTrn <- temp4\n\n# Define max_features\ntokenizer <- text_tokenizer(num_words=500000)\ntokenizer %>% fit_text_tokenizer(train[[\"all_text\"]]) # Insert data here.\n#print(tokenizer$word_counts)\n\nmax_len <- max(train[[\"exposure\"]])\nvocab_size = length(tokenizer$word_index)+1\n\nset.seed(100)\nrand_sample = runif(nrow(train)) < 0.8\n\ntemp <- train\ntrain <- temp[rand_sample]\nvalid <- temp[!rand_sample]\n\ntemp2 <- yTrn\nyTrn <- to_categorical(temp2[rand_sample])\nyVal <- to_categorical(temp2[!rand_sample])\n\nkeras_train <- texts_to_sequences(tokenizer, train[[\"all_text\"]])\npad_train <- pad_sequences(keras_train, maxlen=max_len, padding=\"post\")\n\nkeras_valid <- texts_to_sequences(tokenizer, valid[[\"all_text\"]])\npad_valid <- pad_sequences(keras_valid, maxlen=max_len, padding=\"post\")\n\n\nf1 <- function(y_true, y_pred){\n    y_pred = k_round(y_pred)\n    tp = k_sum(k_cast(y_true*y_pred, 'float'), axis=0)\n    tn = k_sum(k_cast((1-y_true)*(1-y_pred), 'float'), axis=0)\n    fp = k_sum(k_cast((1-y_true)*y_pred, 'float'), axis=0)\n    fn = k_sum(k_cast(y_true*(1-y_pred), 'float'), axis=0)\n\n    p = tp / (tp + fp + k_epsilon())\n    r = tp / (tp + fn + k_epsilon())\n\n    f1 = 2*p*r / (p+r+k_epsilon())\n    return(k_mean(f1))\n}\nf1_loss <- function(y_true, y_pred){ \n    tp = k_sum(k_cast(y_true*y_pred, 'float'), axis=0)\n    tn = k_sum(k_cast((1-y_true)*(1-y_pred), 'float'), axis=0)\n    fp = k_sum(k_cast((1-y_true)*y_pred, 'float'), axis=0)\n    fn = k_sum(k_cast(y_true*(1-y_pred), 'float'), axis=0)\n\n    p = tp / (tp + fp + k_epsilon())\n    r = tp / (tp + fn + k_epsilon())\n\n    f1 = 2*p*r / (p+r+k_epsilon())\n    #f1 = tf.where(tf.is_nan(f1), k_zeros_like(f1), f1)\n\n    return(1 - k_mean(f1))\n}\nr_f1 <- function(y_true,y_pred){\n    tp = sum(y_true*y_pred)\n    tn = sum((1-y_true)*(1-y_pred))\n    fp = sum((1-y_true)*y_pred)\n    fn = sum(y_true*(1-y_pred))\n    \n    p = tp/(tp + fp)\n    r = tp/(tp + fn)\n    f1 = 2*p*r/(p+r)\n    return(mean(r_f1))\n}\n\n# Try to add a Conv1D, Max Pooling 1D into an LSTM Layer.\n\nnum_epochs <- 10\nsize_of_batches <- 32\nlearning_rate <- 0.02\n#layer_embedding(input_dim=vocab_size, output_dim=300, input_length=max_len) %>%\nprint(vocab_size)\nmodel %>% \n    layer_embedding(output_dim=20, input_dim=vocab_size, input_length=max_len) %>%\n    layer_lstm(64, input_shape=c(max_len,vocab_size,100), kernel_initializer=\"glorot_normal\", unit_forget_bias=TRUE, return_sequences=TRUE) %>%\n    layer_batch_normalization() %>%\n    layer_activation_leaky_relu() %>%\n    layer_dropout(0.5) %>%\n    layer_lstm(32, input_shape=c(max_len,vocab_size,32), kernel_initializer=\"glorot_normal\", unit_forget_bias=TRUE, return_sequences=FALSE) %>%\n    layer_batch_normalization() %>%\n    layer_activation_leaky_relu() %>%\n    layer_dropout(0.5) %>%\n    layer_dense(32, kernel_initializer=\"glorot_normal\") %>%\n    layer_batch_normalization() %>%\n    layer_activation_leaky_relu() %>%\n    layer_dropout(0.5) %>%\n    layer_flatten() %>%\n    layer_dense(10, kernel_initializer=\"glorot_normal\", use_bias=FALSE) %>%\n    layer_activation(\"softmax\") %>%\n    layer_dense(units=2) %>%\n    compile(\n        loss=\"binary_crossentropy\", optimizer=optimizer_adam(lr=learning_rate, clipnorm=1, clipvalue=0.5), metrics=\"accuracy\", weighted_metrics = list(\"binary_crossentropy\")\n    )\n\nhistory = model %>% fit(pad_train, yTrn, epochs=num_epochs, batch_size=size_of_batches, verbose=1, callbacks=list(\n    callback_reduce_lr_on_plateau(monitor=\"binary_crossentropy\",factor=0.1),\n    callback_early_stopping(monitor=\"binary_crossentropy\", min_delta=0.00001, patience=5, mode=\"auto\", restore_best_weights=TRUE)\n))\n#model %>% predict_proba()\nprint(model)\n\nyTrn_pred <- model %>% predict_proba(pad_train)\nyVal_pred <- model %>% predict_proba(pad_valid)\n\nprint(yTrn_pred)\n#print(yTrn_pred)\n#print(yVal_pred)\n#print(head(yTrn))\n#print(head(yVal))\nprint(table(yTrn_pred))\n\nprint(table(yTrn%*%0:1, yTrn_pred))\nprint(table(yVal%*%0:1, yVal_pred))\n\n#print(\"Train:\", r_f1(yTrn,yTrn_pred))\n#print(\"Valid:\", r_f1(yVal,yVal_pred))\n# NOTE: Need to make metric f1_score?\n\n\n    \n\n# Change these to LeakyReLU for test!\n\n\n\n#max_len = max(xTrn.exposure)\n#vocab_size = len(tokenizer.word_index) + 1\n\n#keras_train = tokenizer.texts_to_sequences(xTrn.stems)\n#keras_valid = tokenizer.texts_to_sequences(xVal.stems)\n\n#pad_train = sequence.pad_sequences(keras_train, maxlen=max_len, padding=\"post\")\n#pad_valid = sequence.pad_sequences(keras_valid, maxlen=max_len, padding=\"post\")\n\n#pad_train.reshape(-1, 255, 1)\n#pad_valid.reshape(-1, 255, 1)\n\n\n\n\n\n\n\n\n\n#model.add(LSTM(256, input_shape=(255,1), kernel_initializer=\"glorot_normal\", unit_forget_bias=True, return_sequences=True))\n#model.add(BatchNormalization())\n#model.add(ELU())\n#model.add(Dropout(0.5))\n#model.add(LSTM(64, input_shape=(255,1), kernel_initializer=\"glorot_normal\", unit_forget_bias=True, return_sequences=True))\n#model.add(Dense(64, input_dim=256, kernel_initializer=\"glorot_normal\"))\n#model.add(BatchNormalization())\n#model.add(ELU())\n#model.add(Dropout(0.5))\n#model.add(Dense(64, kernel_initializer=\"glorot_normal\"))\n#model.add(BatchNormalization())\n#model.add(ELU())\n#model.add(Dropout(0.5))\n#model.add(Flatten())\n#model.add(Dense(30, input_dim=64, kernel_initializer=\"glorot_normal\", use_bias=False, activation=\"softmax\"))\n\n\n#print(model.summary())\n# TODO: Consider kernel/activity regularizers.\n# Mainly on LSTM layers.\n#adam = Adam(lr=learning_rate, clipnorm=1, clipvalue=0.5)\n#callback = [EarlyStopping(monitor=\"loss\", min_delta=0.00001, patience=5, mode=\"auto\", restore_best_weights=True)]\n#model.compile(loss='categorical_crossentropy', optimizer=adam, metrics=[\"accuracy\"])\n#history = model.fit(pad_train, yTrn_dummy, epochs=num_epoch, batch_size=32, verbose=1, validation_data=(pad_valid, yVal_dummy))\n\n\n\n#import matplotlib.pyplot as plt\n#epoch_count = range(1, len(history.history['loss']) + 1)\n#plt.plot(epoch_count, history.history['loss'], 'r--')\n#plt.plot(epoch_count, history.history['val_loss'], 'b-')\n#plt.legend(['Training Loss', 'Validation Loss'])\n#plt.xlabel('Epoch')\n#plt.ylabel('Loss')\n#plt.show()\n#plt.plot(epoch_count, history.history['categorical_accuracy'], 'r--')\n#plt.plot(epoch_count, history.history['val_categorical_accuracy'], 'b-')\n#plt.legend(['Training Categorical Accuracy', 'Validation Categorical Accuracy'])\n#plt.xlabel('Epoch')\n#plt.ylabel('Cat Accuracy')\n#plt.show()\n#keras_hold_preds1 = model.predict_proba(pad_hold)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model %>% \n    layer_embedding(output_dim=16, input_dim=vocab_size, input_length=max_len) %>%\n    layer_conv1d() %>%\n    layer_activation_leaky_relu() %>%\n    layer_batch_normalization() %>%\n    layer_spatial_dropout_1d(0.15) %>%\n    layer_max_pooling_1d(3) %>%\n\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(head(pad_train,100))\nprint(head(train[[\"exposure\"]]))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"R","language":"R","name":"ir"},"language_info":{"mimetype":"text/x-r-source","name":"R","pygments_lexer":"r","version":"3.4.2","file_extension":".r","codemirror_mode":"r"}},"nbformat":4,"nbformat_minor":4}