## So I had some problems loading pretrained embeddings fast & memory friendly.
## that's why i did some little hacking and used reticulate inside R to load and match
## embeddings. The Python-Scripts are copy pasted from public kernels.
## have fun! 
## Cheers

library(tidyverse)
library(data.table)
library(Matrix)
library(reticulate)
library(keras)
library(tictoc)

tr <- fread('../input/train.csv', data.table = FALSE, encoding = "UTF-8")

maxlen <- as.integer(15)
max_words <- as.integer(5000)
emb_dim <- as.integer(300)

tr <- tr %>% 
  mutate(question_text = ifelse(is.na(question_text),"_na_",question_text),
         question_text = str_replace_all(question_text, "\\s+", " ")) ## remove multiple ws)

tokenizer <- text_tokenizer(num_words = max_words) %>% 
  fit_text_tokenizer(tr$question_text)

word_idx <- tokenizer$word_index

## write some python code
read_emb <- 
"
import io
import os
import time
import numpy as np 

def load_emb(word_index, max_features, embed_size, type = 'glove'):
\tprint('.... reading EMB ', type)
\tif(type == 'glove'):
\t\tEMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'
\tif(type == 'fast_text'):
\t\tEMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'
\tif(type == 'para'):
\t\tEMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'

\tdef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')
\tif(type == 'glove'):
\t\tembeddings_index = dict(get_coefs(*o.split(' ')) for o in io.open(EMBEDDING_FILE, errors='ignore'))
\tif(type == 'fast_text'):
\t\tembeddings_index = dict(get_coefs(*o.split(' ')) for o in io.open(EMBEDDING_FILE, encoding='utf8', errors='ignore') if len(o)>100)
\tif(type == 'para'):
\t\tembeddings_index = dict(get_coefs(*o.split(' ')) for o in io.open(EMBEDDING_FILE, encoding='utf8', errors='ignore') if len(o)>100)

\tall_embs = np.stack(embeddings_index.values())
\temb_mean,emb_std = all_embs.mean(), all_embs.std()
\tembed_size = all_embs.shape[1]
\tprint('reading done')

\tnb_words = min(max_features, len(word_index.items()))
\tembedding_matrix = np.zeros((nb_words+1, embed_size))   #embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words+1, embed_size))

\tfor word, i in word_index.items():
\t\tif i >= max_features: continue
\t\tembedding_vector = embeddings_index.get(word)
\t\tif embedding_vector is not None: embedding_matrix[i] = embedding_vector

\treturn embedding_matrix
"
## dump python code
writeLines(read_emb, "load_emb.py")
## source function
reticulate::source_python("load_emb.py")
## use function
tic("load embeddings")
emb <- load_emb(word_idx,as.integer(max_words),as.integer(emb_dim), type = "glove")
toc()

cat("EMB DIM: ", dim(emb), " ...\n")
print(head(emb[,1:5]))

cat("have fun!", "... \n")