{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# CheXpert Binary Classifier ","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install keras\n","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:56:06.329468Z","iopub.execute_input":"2021-10-14T15:56:06.329781Z","iopub.status.idle":"2021-10-14T15:56:16.154699Z","shell.execute_reply.started":"2021-10-14T15:56:06.329701Z","shell.execute_reply":"2021-10-14T15:56:16.153886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"!pip install --upgrade tensorflow\n!pip install --upgrade tensorflow-gpu","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:56:20.809212Z","iopub.execute_input":"2021-10-14T15:56:20.809487Z","iopub.status.idle":"2021-10-14T15:57:56.726294Z","shell.execute_reply.started":"2021-10-14T15:56:20.809455Z","shell.execute_reply":"2021-10-14T15:57:56.725463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:57:56.729696Z","iopub.execute_input":"2021-10-14T15:57:56.729927Z","iopub.status.idle":"2021-10-14T15:57:58.621784Z","shell.execute_reply.started":"2021-10-14T15:57:56.729900Z","shell.execute_reply":"2021-10-14T15:57:58.621065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tf-keras-vis","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:57:58.622919Z","iopub.execute_input":"2021-10-14T15:57:58.623172Z","iopub.status.idle":"2021-10-14T15:58:06.206307Z","shell.execute_reply.started":"2021-10-14T15:57:58.623139Z","shell.execute_reply":"2021-10-14T15:58:06.205458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### import libraries\nimport tensorflow as tf\nfrom tensorflow import keras\n# import keras\nfrom keras_preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Dense, Activation, Flatten, Dropout, BatchNormalization\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D\nfrom tensorflow.keras import regularizers, optimizers\nfrom tensorflow.keras.models import Sequential\n\n#from keras.applications.densenet import DenseNet121\nfrom tensorflow.keras.applications import  ResNet50\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.models import load_model\n# from keras.utils.vis_utils import plot_model\nfrom tensorflow.keras.utils import to_categorical\n\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\n\nimport matplotlib\nmatplotlib.use(\"Agg\") # set the matplotlib backend so figures can be saved in the background\n \nfrom sklearn.metrics import classification_report\nimport matplotlib.pyplot as plt\n\n#import pydotplus\n#import pydot as pyd\n#from keras.utils.vis_utils import model_to_dot\n#keras.utils.vis_utils.pydot = pyd","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:58:06.210336Z","iopub.execute_input":"2021-10-14T15:58:06.210560Z","iopub.status.idle":"2021-10-14T15:58:06.938538Z","shell.execute_reply.started":"2021-10-14T15:58:06.210532Z","shell.execute_reply":"2021-10-14T15:58:06.937545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dtrain=pd.read_csv(\"../input/csv-tr/train.csv\")\n\n\n# dtrain = dtrain.drop([\"Sex\", \"Age\", \"Frontal/Lateral\", \"AP/PA\"], axis=1)\nprint(dtrain.shape)\ndtrain.describe().transpose()\ndtrain.head()\n\n\n","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:58:06.939947Z","iopub.execute_input":"2021-10-14T15:58:06.940276Z","iopub.status.idle":"2021-10-14T15:58:07.085802Z","shell.execute_reply.started":"2021-10-14T15:58:06.940237Z","shell.execute_reply":"2021-10-14T15:58:07.085150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split data into train/valid/test, use 10% of total data for validation and testing\ndtrain = dtrain.sample(frac=1)\ndvalid_size = round(0.1*dtrain.shape[0])\ndtest_size = dvalid_size\ndtr = dtrain[0:dtrain.shape[0]-dvalid_size-dtest_size+1]\ndv = dtrain[dtrain.shape[0]-dvalid_size-dtest_size:dtrain.shape[0]-dvalid_size+1]\ndte = dtrain[dtrain.shape[0]-dvalid_size:dtrain.shape[0]+1]\n#Just for understanding\nprint(dtr.shape)\nprint(dv.shape)\nprint(dte.shape)\n#dtr.describe()\n#dv.describe()\n#dte.describe().transpose()\n","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:58:07.087180Z","iopub.execute_input":"2021-10-14T15:58:07.087422Z","iopub.status.idle":"2021-10-14T15:58:07.100557Z","shell.execute_reply.started":"2021-10-14T15:58:07.087392Z","shell.execute_reply":"2021-10-14T15:58:07.099748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(dtr)","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:58:07.113205Z","iopub.execute_input":"2021-10-14T15:58:07.113667Z","iopub.status.idle":"2021-10-14T15:58:07.123206Z","shell.execute_reply.started":"2021-10-14T15:58:07.113633Z","shell.execute_reply":"2021-10-14T15:58:07.121834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #This piece of code is only for kaggle it will add path before each file for kaggle directory\n# dtr.Path = r'../input/chexpert/' + dtr.Path\n# dv.Path = r'../input/chexpert/' + dv.Path\n# dte.Path = r'../input/chexpert/' + dte.Path\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(dtr)","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:58:07.124528Z","iopub.execute_input":"2021-10-14T15:58:07.124765Z","iopub.status.idle":"2021-10-14T15:58:07.134237Z","shell.execute_reply.started":"2021-10-14T15:58:07.124743Z","shell.execute_reply":"2021-10-14T15:58:07.133332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### data generation for Keras \ntrain_datagen=ImageDataGenerator(rescale=1./255)\ntest_datagen=ImageDataGenerator(rescale=1./255.)\nvalid_datagen=ImageDataGenerator(rescale=1./255.)\n\ntarget_size = (224,224)\n#target_size = (299,299)\n#target_size = (75,75)\ntrain_generator=train_datagen.flow_from_dataframe(dataframe=dtr, directory=None , x_col=\"paths\", y_col=list(dtr.columns[1:3]), class_mode=\"other\", drop_duplicates = False, target_size=target_size, batch_size=32)\nvalid_generator=valid_datagen.flow_from_dataframe(dataframe=dv, directory=None, x_col=\"paths\", y_col=list(dv.columns[1:3]), class_mode=\"other\", drop_duplicates = False, target_size=target_size, batch_size=32)\ntest_generator=test_datagen.flow_from_dataframe(dataframe=dte, directory=None, x_col=\"paths\", y_col=list(dte.columns[1:3]), class_mode=\"other\", drop_duplicates = False, target_size=target_size, shuffle = False, batch_size=1)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:58:07.137027Z","iopub.execute_input":"2021-10-14T15:58:07.137491Z","iopub.status.idle":"2021-10-14T15:59:25.802434Z","shell.execute_reply.started":"2021-10-14T15:58:07.137457Z","shell.execute_reply":"2021-10-14T15:59:25.801694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install vit_keras","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:59:25.803727Z","iopub.execute_input":"2021-10-14T15:59:25.804178Z","iopub.status.idle":"2021-10-14T15:59:33.516881Z","shell.execute_reply.started":"2021-10-14T15:59:25.804140Z","shell.execute_reply":"2021-10-14T15:59:33.516106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model architecture design/selection\n# create the base pre-trained model\nfrom vit_keras import vit, utils\n\nimage_size = 224\nbase_model = vit.vit_l16(\n    image_size=image_size,\n    activation='sigmoid',\n    pretrained=True,\n    include_top=False,\n    pretrained_top=False\n)\n","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:59:33.519906Z","iopub.execute_input":"2021-10-14T15:59:33.520131Z","iopub.status.idle":"2021-10-14T15:59:59.001106Z","shell.execute_reply.started":"2021-10-14T15:59:33.520103Z","shell.execute_reply":"2021-10-14T15:59:59.000338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #initialize ResNet50 base CNN            \n# from keras.applications import ResNet50\n\n# base_model = ResNet50(include_top = False, # Remove classification head\n#                  weights = 'imagenet',\n#                  input_shape = (224, 224, 3)\n#                    )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# add a global spatial average pooling layer\nx = base_model.output\n#x = GlobalAveragePooling2D()(x)\n# let's add a fully-connected layer\nx = Dense(128, activation='relu')(x)\nx = Dropout(0.25)(x)\n# and a logistic layer\npredictions = Dense(2, activation='sigmoid')(x)\n#predictions = base_model.output\n\n# this is the model we will train\nmodel_F = Model(inputs=base_model.input, outputs=predictions)","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:59:59.002448Z","iopub.execute_input":"2021-10-14T15:59:59.002729Z","iopub.status.idle":"2021-10-14T15:59:59.032866Z","shell.execute_reply.started":"2021-10-14T15:59:59.002696Z","shell.execute_reply":"2021-10-14T15:59:59.032149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# first: train only the top layers (which were randomly initialized)\n# i.e. freeze all layers in base_model layers\nfor layer in base_model.layers:\n    layer.trainable = False\n\n# For Fine-tuning which is an optional step set this layer.trainable to FALSE and compile model \n# again and set training epochs. Note: change polts and file name to avoid over writing","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:59:59.053444Z","iopub.execute_input":"2021-10-14T15:59:59.053640Z","iopub.status.idle":"2021-10-14T15:59:59.072538Z","shell.execute_reply.started":"2021-10-14T15:59:59.053618Z","shell.execute_reply":"2021-10-14T15:59:59.071907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# compile the model (should be done *after* setting layers to non-trainable)\n\nadam = keras.optimizers.Adam(lr=0.0001, beta_1=0.9, beta_2=0.999, epsilon=None, decay=0.0, amsgrad=False)\nmodel_F.compile(optimizer= adam, loss='binary_crossentropy', metrics=['accuracy'])\n\n#plot_model(model, to_file='model_plot.png', show_shapes=True, show_layer_names=True)\n#model_F.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:59:59.073722Z","iopub.execute_input":"2021-10-14T15:59:59.073992Z","iopub.status.idle":"2021-10-14T15:59:59.094183Z","shell.execute_reply.started":"2021-10-14T15:59:59.073960Z","shell.execute_reply":"2021-10-14T15:59:59.093403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### fit model \nnum_epochs = 50\nSTEP_SIZE_TRAIN=train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\nSTEP_SIZE_TEST=test_generator.n//test_generator.batch_size\nmodel_H = model_F.fit(train_generator,\n                    steps_per_epoch=STEP_SIZE_TRAIN,\n                    validation_data=valid_generator,\n                    validation_steps=STEP_SIZE_VALID,\n                    epochs=num_epochs)\n#save model\nmodel_F.save(\"MGMT_1_model_ViTl16_Full_Sample.h5\") #change name according to model","metadata":{"execution":{"iopub.status.busy":"2021-10-14T15:59:59.096421Z","iopub.execute_input":"2021-10-14T15:59:59.096977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert the history.history dict to a pandas DataFrame:     \nhist_df = pd.DataFrame(model_H.history) \nprint(hist_df.head())\n\n# or save to csv: \nhist_csv_file = 'vit_history.csv' #Change name according to model\nwith open(hist_csv_file, mode='w') as f:\n    hist_df.to_csv(f)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plotting Accuracy and loss\nprint(model_H.history.keys())\nplt.clf()\n#plot model Accuracy and loss on training and validation data\nacc = model_H.history['accuracy']\nval_acc = model_H.history['val_accuracy']\nloss = model_H.history['loss']\nval_loss = model_H.history['val_loss']\n\nepochs = range(1, len(acc) + 1)\n\nplt.plot(epochs, acc, 'b', color='green', label='Training acc')\nplt.plot(epochs, val_acc, 'b', color='red',  label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.grid(True)\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\nplt.savefig('model vit acc.svg', bbox_inches='tight') #change according to model\nplt.clf()\n\n\nplt.plot(epochs, loss, 'b', color='green', label='Training loss')\nplt.plot(epochs, val_loss, 'b', color='red', label='Validation loss')\nplt.title('Training and validation loss')\nplt.grid(True)\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\nplt.savefig('model CNN loss.svg', bbox_inches='tight')#Change according to model\nplt.clf()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEP_SIZE_TRAIN=train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\nSTEP_SIZE_TEST=test_generator.n//test_generator.batch_size\n\n# prediction and performance assessment\ntest_generator.reset()\npred=model_F.predict_generator(test_generator, steps=STEP_SIZE_TEST)\npred_bool = (pred >= 0.5)\n\ny_pred = np.array(pred_bool,dtype =int)\n\ndtest = dte.to_numpy()\ny_true = np.array(dtest[:,1:3],dtype=int)\n\n#print(classification_report(y_true, y_pred,target_names=list(dtr.columns[1:15])))\n\nscore, acc = model_F.evaluate_generator(test_generator, steps=STEP_SIZE_TEST)\nprint('Test score:', score)\nprint('Test accuracy:', acc)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_pred.shape)\nprint(y_true.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train loss & accuracy:' , model_F.evaluate(train_generator))\nprint('\\n')\nprint('Test loss & accuracy:' , model_F.evaluate(test_generator))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test\nfrom sklearn.metrics import accuracy_score, confusion_matrix, plot_confusion_matrix\n#define target for testing\ny_test = test_generator.classes\n#make prediction\nyhat_test = pretrainedCNN_model.predict_classes(test_generator)\n\n#get confusion matrix\ncm = confusion_matrix(y_test, yhat_test)\nprint(cm)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test\n### prediction and performance assessment\ntest_generator.reset()\npred=model_F.predict_generator(test_generator, steps=STEP_SIZE_TEST)\npred_bool = (pred >= 0.5)\n\ny_pred = np.array(pred_bool,dtype =int)\n\ndtest = dte.to_numpy()\ny_true = np.array(dtest[:,1:2],dtype=int)\n\nprint(classification_report(y_true, y_pred,target_names=list(dtr.columns[1:2])))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test\n#code from https://www.kaggle.com/basel99/chest-x-ray-images-cnn-handling-overfitting\n\nimport itertools\ndef plot_confusion_matrix(cm, classes,\n                          normalize = False,\n                          title = 'Confusion matrix',\n                          cmap = plt.cm.Blues):\n    \n    plt.figure(figsize = (6,6))\n    plt.imshow(cm, interpolation = 'nearest', cmap = cmap)\n    plt.title(title)\n    plt.colorbar()\n    plt.grid(b = None)\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation = 90)\n    plt.yticks(tick_marks, classes)\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis = 1)[:, np.newaxis]\n\n    thresh = cm.max() / 2.\n    cm = np.round(cm,2)\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j],\n                 fontsize = 12,\n                 horizontalalignment = \"center\",\n                 color = \"white\" if cm[i, j] > thresh else \"black\")\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    #save\n    plt.savefig('ViT-b16_cm.svg', bbox_inches='tight') #Change name according to model\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test\n#plot confustion matrix\nplot_confusion_matrix(cm, classes = ['NORMAL (Class 0)','ABNORMAL (Class 1)'], normalize = False)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get classification report\nprint('Model: ViT Model', '\\n', classification_report(y_test, yhat_test, target_names = ['NORMAL (Class 0)','ABNORMAL (Class 1)']))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fine tuning\n### <span style='color:Red'> caution It is an optional step that may result in overfitting.  </span> ","metadata":{}},{"cell_type":"code","source":"for layer in base_model.layers:\n    layer.trainable = True","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# compile the model (should be done *after* setting layers to non-trainable)\n\nadam = keras.optimizers.Adam(lr=0.0001, beta_1=0.9, beta_2=0.999, epsilon=None, decay=0.0, amsgrad=False)\nmodel_F.compile(optimizer= adam, loss='binary_crossentropy', metrics=['accuracy'])\n\n#plot_model(model, to_file='model_plot.png', show_shapes=True, show_layer_names=True)\nmodel_F.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### fit model \nnum_epochs = 15\nSTEP_SIZE_TRAIN=train_generator.n//train_generator.batch_size\nSTEP_SIZE_VALID=valid_generator.n//valid_generator.batch_size\nSTEP_SIZE_TEST=test_generator.n//test_generator.batch_size\nmodel_H = model_F.fit(train_generator,\n                    steps_per_epoch=STEP_SIZE_TRAIN,\n                    validation_data=valid_generator,\n                    validation_steps=STEP_SIZE_VALID,\n                    epochs=num_epochs)\n# save model\nmodel_F.save(\"model_ViTB16_Full_Sample.h5\") #change name according to model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert the history.history dict to a pandas DataFrame:     \nhist_df = pd.DataFrame(model_H.history) \nprint(hist_df.head())\n\n# or save to csv: \nhist_csv_file = 'Fine_tuned_CNN_history_finetune.csv' #Change name according to model\nwith open(hist_csv_file, mode='w') as f:\n    hist_df.to_csv(f)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plotting Accuracy and loss\nprint(model_H.history.keys())\nplt.clf()\n#plot model Accuracy and loss on training and validation data\nacc = model_H.history['accuracy']\nval_acc = model_H.history['val_accuracy']\nloss = model_H.history['loss']\nval_loss = model_H.history['val_loss']\n\nepochs = range(1, len(acc) + 1)\n\nplt.plot(epochs, acc, 'b', color='green', label='Training acc')\nplt.plot(epochs, val_acc, 'b', color='red',  label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.grid(True)\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()\nplt.savefig('Fine tuned model CNN acc.svg', bbox_inches='tight') #change according to model\nplt.clf()\n\n\nplt.plot(epochs, loss, 'b', color='green', label='Training loss')\nplt.plot(epochs, val_loss, 'b', color='red', label='Validation loss')\nplt.title('Training and validation loss')\nplt.grid(True)\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()\nplt.savefig('Fine tuned model CNN loss.svg', bbox_inches='tight')#Change according to model\nplt.clf()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Heat map or visualization","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom vit_keras import vit, utils, visualize\nurl = '../input/chexpert/CheXpert-v1.0-small/valid/patient64617/study1/view1_frontal.jpg'\nimage = utils.read(url, image_size)\nattention_map = visualize.attention_map(model=model_F, image=image)\n\n# Plot results\nfig, (ax1, ax2) = plt.subplots(ncols=2)\nax1.axis('off')\nax2.axis('off')\nax1.set_title('Original')\nax2.set_title('Attention Map')\n_ = ax1.imshow(image)\n_ = ax2.imshow(attention_map)\nplt.savefig('attention.png',dpi=300)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Links\n* Model Source code: https://github.com/faustomorales/vit-keras\n* Visiion trnasformer paper:https://arxiv.org/abs/2010.11929\n* The Illustrated Transformer NLP blog: http://jalammar.github.io/illustrated-transformer/\n","metadata":{}}]}