{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-29T02:37:49.163014Z","iopub.execute_input":"2022-07-29T02:37:49.163389Z","iopub.status.idle":"2022-07-29T02:37:49.175013Z","shell.execute_reply.started":"2022-07-29T02:37:49.163361Z","shell.execute_reply":"2022-07-29T02:37:49.173703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\n\nimport re","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.196754Z","iopub.execute_input":"2022-07-29T02:37:49.198233Z","iopub.status.idle":"2022-07-29T02:37:49.202489Z","shell.execute_reply.started":"2022-07-29T02:37:49.198200Z","shell.execute_reply":"2022-07-29T02:37:49.201565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/nlp-getting-started/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.240461Z","iopub.execute_input":"2022-07-29T02:37:49.241449Z","iopub.status.idle":"2022-07-29T02:37:49.272429Z","shell.execute_reply.started":"2022-07-29T02:37:49.241404Z","shell.execute_reply":"2022-07-29T02:37:49.271317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A simple preprocessing for tweets","metadata":{}},{"cell_type":"code","source":"def preprocessing(tweet):\n    # remove stock market tickers like $GE\n    tweet = re.sub(r'\\$\\w*', '', tweet)\n    # remove old style retweet text \"RT\"\n    tweet = re.sub(r'^RT[\\s]+', '', tweet)\n    # remove hyperlinks    \n    tweet = re.sub(r'https?://[^\\s\\n\\r]+', '', tweet)\n    # remove hashtags\n    # only removing the hash # sign from the word\n    tweet = re.sub(r'#', '', tweet)\n    \n    return tweet","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.278206Z","iopub.execute_input":"2022-07-29T02:37:49.278582Z","iopub.status.idle":"2022-07-29T02:37:49.286499Z","shell.execute_reply.started":"2022-07-29T02:37:49.278554Z","shell.execute_reply":"2022-07-29T02:37:49.284604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text'].apply(preprocessing)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.311580Z","iopub.execute_input":"2022-07-29T02:37:49.312667Z","iopub.status.idle":"2022-07-29T02:37:49.403526Z","shell.execute_reply.started":"2022-07-29T02:37:49.312622Z","shell.execute_reply":"2022-07-29T02:37:49.402400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(\n    df['text'].tolist(), \n    df['target'].tolist(),\n    train_size = .8,\n    test_size = .2,\n    shuffle = True,\n    random_state = 42\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.405487Z","iopub.execute_input":"2022-07-29T02:37:49.405808Z","iopub.status.idle":"2022-07-29T02:37:49.417901Z","shell.execute_reply.started":"2022-07-29T02:37:49.405780Z","shell.execute_reply":"2022-07-29T02:37:49.416784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(X_train), len(y_train))\nprint(len(X_test), len(y_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.419366Z","iopub.execute_input":"2022-07-29T02:37:49.419707Z","iopub.status.idle":"2022-07-29T02:37:49.431397Z","shell.execute_reply.started":"2022-07-29T02:37:49.419672Z","shell.execute_reply":"2022-07-29T02:37:49.430432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create vocabulary","metadata":{}},{"cell_type":"code","source":"VOCAB_SIZE = 1000\nencoder = tf.keras.layers.TextVectorization(\n    max_tokens=VOCAB_SIZE)\nencoder.adapt(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.450037Z","iopub.execute_input":"2022-07-29T02:37:49.450474Z","iopub.status.idle":"2022-07-29T02:37:49.861439Z","shell.execute_reply.started":"2022-07-29T02:37:49.450443Z","shell.execute_reply":"2022-07-29T02:37:49.860102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab = np.array(encoder.get_vocabulary())\nvocab[:20]","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.865860Z","iopub.execute_input":"2022-07-29T02:37:49.866203Z","iopub.status.idle":"2022-07-29T02:37:49.879762Z","shell.execute_reply.started":"2022-07-29T02:37:49.866174Z","shell.execute_reply":"2022-07-29T02:37:49.878164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_example = encoder(X_train)[:3].numpy()\nencoded_example","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.881107Z","iopub.execute_input":"2022-07-29T02:37:49.881454Z","iopub.status.idle":"2022-07-29T02:37:49.973663Z","shell.execute_reply.started":"2022-07-29T02:37:49.881425Z","shell.execute_reply":"2022-07-29T02:37:49.972528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for n in range(3):\n  print(\"Original: \", X_train[n])\n  print(\"Round-trip: \", \" \".join(vocab[encoded_example[n]]))\n  print()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.977315Z","iopub.execute_input":"2022-07-29T02:37:49.978631Z","iopub.status.idle":"2022-07-29T02:37:49.986341Z","shell.execute_reply.started":"2022-07-29T02:37:49.978579Z","shell.execute_reply":"2022-07-29T02:37:49.984845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create the model","metadata":{}},{"cell_type":"markdown","source":"## Simple RNN model","metadata":{}},{"cell_type":"code","source":"rnn_model = tf.keras.Sequential([\n    encoder,\n    tf.keras.layers.Embedding(\n        input_dim=len(encoder.get_vocabulary()),\n        output_dim=64,\n        # Use masking to handle the variable sequence lengths\n        mask_zero=True),\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(64)),\n    tf.keras.layers.Dense(64, activation='relu'),\n    tf.keras.layers.Dense(1)\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:49.989192Z","iopub.execute_input":"2022-07-29T02:37:49.989618Z","iopub.status.idle":"2022-07-29T02:37:52.346921Z","shell.execute_reply.started":"2022-07-29T02:37:49.989579Z","shell.execute_reply":"2022-07-29T02:37:52.345689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(rnn_model)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:52.348679Z","iopub.execute_input":"2022-07-29T02:37:52.349028Z","iopub.status.idle":"2022-07-29T02:37:52.454229Z","shell.execute_reply.started":"2022-07-29T02:37:52.348999Z","shell.execute_reply":"2022-07-29T02:37:52.453290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print([layer.supports_masking for layer in rnn_model.layers])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:52.456107Z","iopub.execute_input":"2022-07-29T02:37:52.457495Z","iopub.status.idle":"2022-07-29T02:37:52.464468Z","shell.execute_reply.started":"2022-07-29T02:37:52.457443Z","shell.execute_reply":"2022-07-29T02:37:52.463202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict on a sample text without padding.\n\nsample_text = ('The movie was cool. The animation and the graphics '\n               'were out of this world. I would recommend this movie.')\npredictions = rnn_model.predict(np.array([sample_text]))\nprint(predictions[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:52.466371Z","iopub.execute_input":"2022-07-29T02:37:52.467442Z","iopub.status.idle":"2022-07-29T02:37:55.059872Z","shell.execute_reply.started":"2022-07-29T02:37:52.467405Z","shell.execute_reply":"2022-07-29T02:37:55.058831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict on a sample text with padding\n\npadding = \"the \" * 2000\npredictions = rnn_model.predict(np.array([sample_text, padding]))\nprint(predictions[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:55.060785Z","iopub.execute_input":"2022-07-29T02:37:55.061354Z","iopub.status.idle":"2022-07-29T02:37:55.313013Z","shell.execute_reply.started":"2022-07-29T02:37:55.061325Z","shell.execute_reply":"2022-07-29T02:37:55.311780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compile the model!","metadata":{}},{"cell_type":"code","source":"rnn_model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n              optimizer=tf.keras.optimizers.Adam(1e-4),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:55.317703Z","iopub.execute_input":"2022-07-29T02:37:55.318175Z","iopub.status.idle":"2022-07-29T02:37:55.329645Z","shell.execute_reply.started":"2022-07-29T02:37:55.318142Z","shell.execute_reply":"2022-07-29T02:37:55.328497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Using stack LSTM layers","metadata":{}},{"cell_type":"code","source":"lstm_model = tf.keras.Sequential([\n    encoder,\n    tf.keras.layers.Embedding(len(encoder.get_vocabulary()), 64, mask_zero = True),\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(64, return_sequences = True)),\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(32)),\n    tf.keras.layers.Dense(64, activation = 'relu'),\n    tf.keras.layers.Dropout(.5),\n    tf.keras.layers.Dense(1)\n])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:55.331343Z","iopub.execute_input":"2022-07-29T02:37:55.331731Z","iopub.status.idle":"2022-07-29T02:37:58.868912Z","shell.execute_reply.started":"2022-07-29T02:37:55.331694Z","shell.execute_reply":"2022-07-29T02:37:58.867738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(lstm_model)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:58.870797Z","iopub.execute_input":"2022-07-29T02:37:58.871188Z","iopub.status.idle":"2022-07-29T02:37:58.986767Z","shell.execute_reply.started":"2022-07-29T02:37:58.871150Z","shell.execute_reply":"2022-07-29T02:37:58.985623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compile the model","metadata":{}},{"cell_type":"code","source":"lstm_model.compile(\n    loss = tf.keras.losses.BinaryCrossentropy(from_logits = True),\n    optimizer = tf.keras.optimizers.Adam(1e-4),\n    metrics = ['accuracy']\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:58.988246Z","iopub.execute_input":"2022-07-29T02:37:58.988600Z","iopub.status.idle":"2022-07-29T02:37:59.003579Z","shell.execute_reply.started":"2022-07-29T02:37:58.988568Z","shell.execute_reply":"2022-07-29T02:37:59.002163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training!","metadata":{}},{"cell_type":"code","source":"rnn_history = rnn_model.fit(X_train, y_train, epochs=10,\n                    validation_data=(X_test, y_test),\n                    validation_steps=30)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:37:59.005358Z","iopub.execute_input":"2022-07-29T02:37:59.005682Z","iopub.status.idle":"2022-07-29T02:39:53.943624Z","shell.execute_reply.started":"2022-07-29T02:37:59.005651Z","shell.execute_reply":"2022-07-29T02:39:53.942789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lstm_history = lstm_model.fit(X_train, y_train, epochs = 10,\n                            validation_data = (X_test, y_test),\n                            validation_steps = 30)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:40:12.510655Z","iopub.execute_input":"2022-07-29T02:40:12.511042Z","iopub.status.idle":"2022-07-29T02:43:54.093143Z","shell.execute_reply.started":"2022-07-29T02:40:12.511011Z","shell.execute_reply":"2022-07-29T02:43:54.092344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n\ndef plot_graphs(history, metric):\n  plt.plot(history.history[metric])\n  plt.plot(history.history['val_'+metric], '')\n  plt.xlabel(\"Epochs\")\n  plt.ylabel(metric)\n  plt.legend([metric, 'val_'+metric])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:43:54.095155Z","iopub.execute_input":"2022-07-29T02:43:54.095889Z","iopub.status.idle":"2022-07-29T02:43:54.103339Z","shell.execute_reply.started":"2022-07-29T02:43:54.095827Z","shell.execute_reply":"2022-07-29T02:43:54.101996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The RNN model","metadata":{}},{"cell_type":"code","source":"test_loss, test_acc = rnn_model.evaluate(X_test, y_test)\n\nprint('Test Loss:', test_loss)\nprint('Test Accuracy:', test_acc)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:43:54.105430Z","iopub.execute_input":"2022-07-29T02:43:54.106000Z","iopub.status.idle":"2022-07-29T02:43:54.968561Z","shell.execute_reply.started":"2022-07-29T02:43:54.105968Z","shell.execute_reply":"2022-07-29T02:43:54.967760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nplt.subplot(1, 2, 1)\nplot_graphs(rnn_history, 'accuracy')\nplt.ylim(None, 1)\nplt.subplot(1, 2, 2)\nplot_graphs(rnn_history, 'loss')\nplt.ylim(0, None)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:43:54.970568Z","iopub.execute_input":"2022-07-29T02:43:54.970823Z","iopub.status.idle":"2022-07-29T02:43:55.297910Z","shell.execute_reply.started":"2022-07-29T02:43:54.970796Z","shell.execute_reply":"2022-07-29T02:43:55.296671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The LSTM model","metadata":{}},{"cell_type":"code","source":"test_loss, test_acc = lstm_model.evaluate(X_test, y_test)\n\nprint('Test Loss:', test_loss)\nprint('Test Accuracy:', test_acc)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:43:55.299695Z","iopub.execute_input":"2022-07-29T02:43:55.300254Z","iopub.status.idle":"2022-07-29T02:43:56.765039Z","shell.execute_reply.started":"2022-07-29T02:43:55.300223Z","shell.execute_reply":"2022-07-29T02:43:56.763615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 8))\nplt.subplot(1, 2, 1)\nplot_graphs(lstm_history, 'accuracy')\nplt.ylim(None, 1)\nplt.subplot(1, 2, 2)\nplot_graphs(lstm_history, 'loss')\nplt.ylim(0, None)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:43:56.766723Z","iopub.execute_input":"2022-07-29T02:43:56.767021Z","iopub.status.idle":"2022-07-29T02:43:57.093462Z","shell.execute_reply.started":"2022-07-29T02:43:56.766993Z","shell.execute_reply":"2022-07-29T02:43:57.092399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# Building the test submission\ntest_df = pd.read_csv('../input/nlp-getting-started/test.csv')\n\nX_submit = test_df['text'].tolist()\n\n# model = rnn_model\nmodel = lstm_model","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:44:26.210750Z","iopub.execute_input":"2022-07-29T02:44:26.211136Z","iopub.status.idle":"2022-07-29T02:44:26.233785Z","shell.execute_reply.started":"2022-07-29T02:44:26.211107Z","shell.execute_reply":"2022-07-29T02:44:26.232529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_submit)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:44:28.521729Z","iopub.execute_input":"2022-07-29T02:44:28.522106Z","iopub.status.idle":"2022-07-29T02:44:36.180477Z","shell.execute_reply.started":"2022-07-29T02:44:28.522077Z","shell.execute_reply":"2022-07-29T02:44:36.179594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred[y_pred >= 0] = 1\ny_pred[y_pred < 0] = 0","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:44:36.182111Z","iopub.execute_input":"2022-07-29T02:44:36.182617Z","iopub.status.idle":"2022-07-29T02:44:36.186747Z","shell.execute_reply.started":"2022-07-29T02:44:36.182587Z","shell.execute_reply":"2022-07-29T02:44:36.186049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:44:36.188300Z","iopub.execute_input":"2022-07-29T02:44:36.189512Z","iopub.status.idle":"2022-07-29T02:44:36.202183Z","shell.execute_reply.started":"2022-07-29T02:44:36.189481Z","shell.execute_reply":"2022-07-29T02:44:36.201439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = [int(i[0]) for i in y_pred]","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:44:36.204256Z","iopub.execute_input":"2022-07-29T02:44:36.204967Z","iopub.status.idle":"2022-07-29T02:44:36.212018Z","shell.execute_reply.started":"2022-07-29T02:44:36.204938Z","shell.execute_reply":"2022-07-29T02:44:36.211172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(y_pred), len(test_df['id'].tolist()))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:44:36.213582Z","iopub.execute_input":"2022-07-29T02:44:36.214172Z","iopub.status.idle":"2022-07-29T02:44:36.223623Z","shell.execute_reply.started":"2022-07-29T02:44:36.214133Z","shell.execute_reply":"2022-07-29T02:44:36.222717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame({\n    'id' : test_df['id'].tolist(),\n    'target' : y_pred\n})","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:44:36.224776Z","iopub.execute_input":"2022-07-29T02:44:36.225471Z","iopub.status.idle":"2022-07-29T02:44:36.236612Z","shell.execute_reply.started":"2022-07-29T02:44:36.225439Z","shell.execute_reply":"2022-07-29T02:44:36.235599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:44:38.193594Z","iopub.execute_input":"2022-07-29T02:44:38.193982Z","iopub.status.idle":"2022-07-29T02:44:38.207194Z","shell.execute_reply.started":"2022-07-29T02:44:38.193952Z","shell.execute_reply":"2022-07-29T02:44:38.205848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('/kaggle/working/submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T02:44:42.840186Z","iopub.execute_input":"2022-07-29T02:44:42.841045Z","iopub.status.idle":"2022-07-29T02:44:42.855703Z","shell.execute_reply.started":"2022-07-29T02:44:42.841012Z","shell.execute_reply":"2022-07-29T02:44:42.854281Z"},"trusted":true},"execution_count":null,"outputs":[]}]}