{"cells":[{"metadata":{"_uuid":"a502fa39d2de3773bf45935f326a7c88fa4fe7d8"},"cell_type":"markdown","source":"In this kernel I try to convert the Pytorch Starter Kernel (https://www.kaggle.com/hung96ad/pytorch-starter) to the Fastai framework. I didn't use their textcleaning but implemented the default tokenization of FastAI. "},{"metadata":{"trusted":true,"_uuid":"15b4f35d14b69843e9075fb21b78865c28666696"},"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true},"cell_type":"code","source":"import re\nimport time\nimport gc\nimport random\nimport os\n\nimport numpy as np\nimport pandas as pd\n\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n#from sklearn.model_selection import GridSearchCV, StratifiedKFold\nfrom sklearn.metrics import f1_score, roc_auc_score\n\n\nimport torch\nimport torch.nn as nn\nimport torch.utils.data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ad743610aa7bb3f3af516a4c9a909e30378065f7"},"cell_type":"code","source":"\nfrom fastai.text import *","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5cca10128189b2622460262af12b163dda1dc81e"},"cell_type":"code","source":"import fastai\nfastai.__version__","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"def seed_torch(seed=1029):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"85ee6b26944da2cd972c5038fa63c78b763184d3","trusted":true},"cell_type":"code","source":"embed_size = 300 # how big is each word vector\nmax_features = 60000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 72 # max number of words in a question to use\n\nbatch_size = 1536\ntrain_epochs = 8\n\nSEED = 1029","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9fa4b3016d2edf2c14340de1f23988b086850851","trusted":true},"cell_type":"code","source":"def load_glove(word_index):\n    EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n            \n    return embedding_matrix \n\ndef load_para(word_index):\n    EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    # word_index = tokenizer.word_index\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    \n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6be9d1b3233250ff3bba1bd749c3b3b2eac75f3d"},"cell_type":"markdown","source":"# Datablock"},{"metadata":{"trusted":true,"_uuid":"0e8692168a0cee5c732ae76061efc1b970b1ae2f"},"cell_type":"code","source":"path = Path('../input/')\nMODEL_PATH = Path('../working/')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e284cb47d16dbb5c5558838914ee50a616932b5"},"cell_type":"code","source":"\ndf = pd.read_csv(path/'train.csv',)\ndf_test = pd.read_csv(path/'test.csv',)\ndf[df['target']==1].head()\n\n#rename columns\ndf.rename(columns= {'question_text' : 'text', 'target' : 'label'}, inplace=True)\ndf_test.rename(columns= {'question_text' : 'text'}, inplace=True)\n\n#select columns\ndf = df[['text', 'label']]\n\n#save the cleaned file\ndf.to_csv(f\"clean.csv\", index=False, header=True)\ndf_test.to_csv(f\"clean_test.csv\", index=False, header=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6a65f4ff778e9f0134cf116bb9f90fecd5a4a9ad"},"cell_type":"code","source":"bs =48","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1bf11271fd81d43d9c3f2a2c3418a08eef7cc465"},"cell_type":"code","source":"txt_proc = [\n    TokenizeProcessor(tokenizer=Tokenizer(lang='en'), mark_fields=True),\n    NumericalizeProcessor(min_freq=2, max_vocab=60000)\n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"473c41a05a811de0c827d8c21788e51281aadac2"},"cell_type":"code","source":"src = (TextList.from_df(df, processor=txt_proc)\n                .random_split_by_pct(0.1) \n                .label_from_df(1)                \n              ) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c07baf16c6d0053a0f63873adae4b1fcda6cc20b"},"cell_type":"code","source":"data = src.databunch(bs= bs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2da2ae287aed494ab0d517a0de5b5015e5b1fd65"},"cell_type":"code","source":"data.save('tmp_clas')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"762cab1b53246e8aeb9793aa17d82720f13e3729"},"cell_type":"code","source":"data = TextDataBunch.load(MODEL_PATH, 'tmp_clas', bs=bs)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c97688982af24c34992f01bf89447e78ba982081"},"cell_type":"code","source":"word_index = data.vocab.stoi\nprint(type(word_index)), print(len(word_index))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d799741a7c93f7a71322e7928331da87e27874f"},"cell_type":"code","source":"word_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df285dea4f8402c075c0ddd6157b6a1071769339"},"cell_type":"code","source":"start_time = time.time()\n\nembedding_matrix_1 = load_glove(word_index)\nembedding_matrix_2 = load_para(word_index)\n\n\n\ntotal_time = (time.time() - start_time) / 60\nprint(\"Took {:.2f} minutes\".format(total_time))\n\nembedding_matrix = np.mean([embedding_matrix_1, embedding_matrix_2], axis=0)\n# embedding_matrix = np.concatenate((embedding_matrix_1, embedding_matrix_2), axis=1)\nprint(np.shape(embedding_matrix))\n\ndel embedding_matrix_1, embedding_matrix_2\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af76df5311ed1ce7d65291adaff016255160af31"},"cell_type":"code","source":"np.save('tmp_embedding.npy', embedding_matrix)\ntype(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c3444902b0b56b9aeb38632d11cf4519bfc7a88"},"cell_type":"code","source":"embedding_matrix = np.load('tmp_embedding.npy')\nembedding_matrix.shape, embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c3cc36a01404a442e1350cc5db830e1e30bb7640"},"cell_type":"markdown","source":"# Model"},{"metadata":{"trusted":true,"_uuid":"f653b4aa968f1a94b3879e6b9066dec5b78e59d9"},"cell_type":"code","source":"data.train_ds.x","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b6c179e5478f1ac92b69020fb875bf3d8154bff3","trusted":true},"cell_type":"code","source":"def sigmoid(x):\n    return 1 / (1 + np.exp(-x))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e78df8d5ca5898ada4e6bf99132c00640493d351","trusted":true},"cell_type":"code","source":"class Attention(nn.Module):\n    def __init__(self, feature_dim, step_dim, bias=True, **kwargs):\n        super(Attention, self).__init__(**kwargs)\n        \n        self.supports_masking = True\n\n        self.bias = bias\n        self.feature_dim = feature_dim\n        self.step_dim = step_dim\n        self.features_dim = 0\n        \n        weight = torch.zeros(feature_dim, 1)\n        nn.init.xavier_uniform_(weight)\n        self.weight = nn.Parameter(weight)\n        \n        if bias:\n            self.b = nn.Parameter(torch.zeros(step_dim))\n        \n    def forward(self, x, mask=None):\n        feature_dim = self.feature_dim\n        step_dim = self.step_dim\n\n        eij = torch.mm(\n            x.contiguous().view(-1, feature_dim), \n            self.weight\n        ).view(-1, step_dim)\n        \n        if self.bias:\n            eij = eij + self.b\n            \n        eij = torch.tanh(eij)\n        a = torch.exp(eij)\n        \n        if mask is not None:\n            a = a * mask\n\n        a = a / torch.sum(a, 1, keepdim=True) + 1e-10\n\n        weighted_input = x * torch.unsqueeze(a, -1)\n        return torch.sum(weighted_input, 1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"71eeb24155b28b82e08b89becf8ba2cf3e207d67","trusted":true},"cell_type":"code","source":"class NeuralNet(nn.Module):\n    def __init__(self):\n        super(NeuralNet, self).__init__()\n        \n        hidden_size = 60\n        \n        self.embedding = nn.Embedding(max_features, embed_size)\n        self.embedding.weight = nn.Parameter(torch.tensor(embedding_matrix, dtype=torch.float32))\n        self.embedding.weight.requires_grad = False\n        \n        self.embedding_dropout = nn.Dropout2d(0.1)\n        self.lstm = nn.GRU(embed_size, hidden_size, bidirectional=True, batch_first=True)\n        self.gru = nn.GRU(hidden_size*2, hidden_size, bidirectional=True, batch_first=True)\n        \n        self.lstm_attention = Attention(hidden_size*2, maxlen)\n        self.gru_attention = Attention(hidden_size*2, maxlen)\n        \n        self.linear = nn.Linear(480, 16)\n        self.relu = nn.ReLU()\n        self.dropout = nn.Dropout(0.1)\n        self.out = nn.Linear(16, 1)\n        \n    def forward(self, x):\n        h_embedding = self.embedding(x)\n        h_embedding = torch.squeeze(self.embedding_dropout(torch.unsqueeze(h_embedding, 0)))\n        \n        h_lstm, _ = self.lstm(h_embedding)\n        h_gru, _ = self.gru(h_lstm)\n        \n        h_lstm_atten = self.lstm_attention(h_lstm)\n        h_gru_atten = self.gru_attention(h_gru)\n        \n        avg_pool = torch.mean(h_gru, 1)\n        max_pool, _ = torch.max(h_gru, 1)\n        \n        conc = torch.cat((h_lstm_atten, h_gru_atten, avg_pool, max_pool), 1)\n        conc = self.relu(self.linear(conc))\n        conc = self.dropout(conc)\n        out = self.out(conc)\n        \n        return out","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"451e1d80f90d56ba689f70599426384695442580"},"cell_type":"code","source":"print(max_features)\nprint(embed_size)\nprint(embedding_matrix.shape)\nprint(maxlen)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db04d84f47f0a0f384efc8868da577d0d74f870f"},"cell_type":"code","source":"model = NeuralNet()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a36bb16a28f725eb0ae82ee8f74d9ba6cfc24d1f"},"cell_type":"code","source":"gc.collect()\n\nmodel.parameters","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aff9a9c2bb9dbaa6016e357ca7c43f3482692c75"},"cell_type":"code","source":"att = Attention()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"edd136c45d1381a4f3b128336380897ce04b3053"},"cell_type":"code","source":"\nlearn_clas = Learner(data, model, metrics=[sigmoid])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1bdb97bd45b2b1c4b80f68d2b09fbeab958e782b"},"cell_type":"code","source":"learn_clas.fit_one_cycle(1, 1e-3)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"96f3f95ef41c92f49b74415784e5b2b0be83cc79"},"cell_type":"markdown","source":"# Todo\nTrying to figure out how to implement the custom model into the fastai framework"},{"metadata":{"_uuid":"2ad4bf11f982e199ff5d64004eeedd9f3d3d454c","trusted":false},"cell_type":"code","source":"def threshold_search(y_true, y_proba):\n    best_threshold = 0\n    best_score = 0\n    for threshold in tqdm([i * 0.01 for i in range(100)]):\n        score = f1_score(y_true=y_true, y_pred=y_proba > threshold)\n        if score > best_score:\n            best_threshold = threshold\n            best_score = score\n    search_result = {'threshold': best_threshold, 'f1': best_score}\n    return search_result","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1f80aa8c099cb76eede9c0c5db859a835fc9d1f9","trusted":false},"cell_type":"code","source":"train_preds = np.zeros((len(train_X)))\ntest_preds = np.zeros((len(test_X)))\n\nseed_torch(SEED)\n\nx_test_cuda = torch.tensor(test_X, dtype=torch.long).cuda()\ntest = torch.utils.data.TensorDataset(x_test_cuda)\ntest_loader = torch.utils.data.DataLoader(test, batch_size=batch_size, shuffle=False)\n\nfor i, (train_idx, valid_idx) in enumerate(splits):\n    x_train_fold = torch.tensor(train_X[train_idx], dtype=torch.long).cuda()\n    y_train_fold = torch.tensor(train_y[train_idx, np.newaxis], dtype=torch.float32).cuda()\n    x_val_fold = torch.tensor(train_X[valid_idx], dtype=torch.long).cuda()\n    y_val_fold = torch.tensor(train_y[valid_idx, np.newaxis], dtype=torch.float32).cuda()\n    \n    model = NeuralNet()\n    model.cuda()\n    \n    loss_fn = torch.nn.BCEWithLogitsLoss(reduction=\"sum\")\n    optimizer = torch.optim.Adam(model.parameters())\n    \n    train = torch.utils.data.TensorDataset(x_train_fold, y_train_fold)\n    valid = torch.utils.data.TensorDataset(x_val_fold, y_val_fold)\n    \n    train_loader = torch.utils.data.DataLoader(train, batch_size=batch_size, shuffle=True)\n    valid_loader = torch.utils.data.DataLoader(valid, batch_size=batch_size, shuffle=False)\n    \n    print(f'Fold {i + 1}')\n    \n    for epoch in range(train_epochs):\n        start_time = time.time()\n        \n        model.train()\n        avg_loss = 0.\n        for x_batch, y_batch in tqdm(train_loader, disable=True):\n            y_pred = model(x_batch)\n            loss = loss_fn(y_pred, y_batch)\n            optimizer.zero_grad()\n            loss.backward()\n            optimizer.step()\n            avg_loss += loss.item() / len(train_loader)\n        \n        model.eval()\n        valid_preds_fold = np.zeros((x_val_fold.size(0)))\n        test_preds_fold = np.zeros(len(test_X))\n        avg_val_loss = 0.\n        for i, (x_batch, y_batch) in enumerate(valid_loader):\n            y_pred = model(x_batch).detach()\n            avg_val_loss += loss_fn(y_pred, y_batch).item() / len(valid_loader)\n            valid_preds_fold[i * batch_size:(i+1) * batch_size] = sigmoid(y_pred.cpu().numpy())[:, 0]\n        \n        elapsed_time = time.time() - start_time \n        print('Epoch {}/{} \\t loss={:.4f} \\t val_loss={:.4f} \\t time={:.2f}s'.format(\n            epoch + 1, train_epochs, avg_loss, avg_val_loss, elapsed_time))\n        \n    for i, (x_batch,) in enumerate(test_loader):\n        y_pred = model(x_batch).detach()\n\n        test_preds_fold[i * batch_size:(i+1) * batch_size] = sigmoid(y_pred.cpu().numpy())[:, 0]\n\n    train_preds[valid_idx] = valid_preds_fold\n    test_preds += test_preds_fold / len(splits)    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f0c7655686bd19130f7f2c8074b2ec44b1451848","trusted":false},"cell_type":"code","source":"search_result = threshold_search(train_y, train_preds)\nsearch_result","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c975abbb27b1fdd85403044d521ee0eccbb7da1c"},"cell_type":"markdown","source":"# Submission"},{"metadata":{"_uuid":"c3581f74ae694eb07182e2a23c9db8f01a78f1ba","trusted":false},"cell_type":"code","source":"sub = pd.read_csv('../input/sample_submission.csv')\nsub.prediction = test_preds > search_result['threshold']\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}