{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-14T04:26:22.780291Z","iopub.execute_input":"2022-08-14T04:26:22.781347Z","iopub.status.idle":"2022-08-14T04:26:22.787189Z","shell.execute_reply.started":"2022-08-14T04:26:22.781299Z","shell.execute_reply":"2022-08-14T04:26:22.786095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/nlp-getting-started/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:22.907450Z","iopub.execute_input":"2022-08-14T04:26:22.907729Z","iopub.status.idle":"2022-08-14T04:26:22.944987Z","shell.execute_reply.started":"2022-08-14T04:26:22.907703Z","shell.execute_reply":"2022-08-14T04:26:22.944055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:42:06.321326Z","iopub.execute_input":"2022-08-14T04:42:06.322323Z","iopub.status.idle":"2022-08-14T04:42:06.331703Z","shell.execute_reply.started":"2022-08-14T04:42:06.322285Z","shell.execute_reply":"2022-08-14T04:42:06.330666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df[['text', 'target']]\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:22.973447Z","iopub.execute_input":"2022-08-14T04:26:22.974021Z","iopub.status.idle":"2022-08-14T04:26:22.985653Z","shell.execute_reply.started":"2022-08-14T04:26:22.973982Z","shell.execute_reply":"2022-08-14T04:26:22.984279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = df['text']\nlabels = df['target']","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:23.019451Z","iopub.execute_input":"2022-08-14T04:26:23.020037Z","iopub.status.idle":"2022-08-14T04:26:23.025799Z","shell.execute_reply.started":"2022-08-14T04:26:23.020009Z","shell.execute_reply":"2022-08-14T04:26:23.024902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_ds = tf.data.Dataset.from_tensor_slices((features, labels))","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:23.064594Z","iopub.execute_input":"2022-08-14T04:26:23.065136Z","iopub.status.idle":"2022-08-14T04:26:23.076692Z","shell.execute_reply.started":"2022-08-14T04:26:23.065109Z","shell.execute_reply":"2022-08-14T04:26:23.075716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_split = int(df.shape[0] * .2)\nval_ds = text_ds.take(test_split)\ntrain_ds = text_ds.skip(test_split)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:23.112991Z","iopub.execute_input":"2022-08-14T04:26:23.114806Z","iopub.status.idle":"2022-08-14T04:26:23.119930Z","shell.execute_reply.started":"2022-08-14T04:26:23.114779Z","shell.execute_reply":"2022-08-14T04:26:23.118884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BUFFER_SIZE = df.shape[0]\nAUTOTUNE = tf.data.AUTOTUNE\nBATCH_SIZE = 32\n\ntrain_ds = train_ds.shuffle(BUFFER_SIZE).batch(BATCH_SIZE).prefetch(AUTOTUNE)\nval_ds = text_ds.batch(BATCH_SIZE).prefetch(AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:23.138522Z","iopub.execute_input":"2022-08-14T04:26:23.139057Z","iopub.status.idle":"2022-08-14T04:26:23.147018Z","shell.execute_reply.started":"2022-08-14T04:26:23.139021Z","shell.execute_reply":"2022-08-14T04:26:23.146124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport string\n\ndef standardize(input_data):\n    lower_case = tf.strings.lower(input_data)\n    stripped_html = tf.strings.regex_replace(lower_case, 'http\\S+', '')\n    punct_removed = tf.strings.regex_replace(stripped_html, \n                                    '[%s]' % re.escape(string.punctuation), '')\n    return tf.strings.regex_replace(punct_removed,\n                                   '[^\\x00-\\x7f]', '')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:23.151943Z","iopub.execute_input":"2022-08-14T04:26:23.152948Z","iopub.status.idle":"2022-08-14T04:26:23.159187Z","shell.execute_reply.started":"2022-08-14T04:26:23.152913Z","shell.execute_reply":"2022-08-14T04:26:23.158110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'text: {df[\"text\"][200]}')\nprint(f'standardized text: {standardize(df[\"text\"][200]).numpy()}')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:23.162224Z","iopub.execute_input":"2022-08-14T04:26:23.162766Z","iopub.status.idle":"2022-08-14T04:26:23.171207Z","shell.execute_reply.started":"2022-08-14T04:26:23.162702Z","shell.execute_reply":"2022-08-14T04:26:23.170068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"VOCAB_SIZE = 10_000\nMAX_LEN = 150\n\nvectorization_layer = tf.keras.layers.TextVectorization(max_tokens=VOCAB_SIZE, \n                                                        standardize=standardize,\n                                                        output_mode='int',\n                                                       output_sequence_length=MAX_LEN)\nvectorization_layer.adapt(train_ds.map(lambda text, label: text))","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:23.217693Z","iopub.execute_input":"2022-08-14T04:26:23.218228Z","iopub.status.idle":"2022-08-14T04:26:23.630149Z","shell.execute_reply.started":"2022-08-14T04:26:23.218193Z","shell.execute_reply":"2022-08-14T04:26:23.629131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab = np.array(vectorization_layer.get_vocabulary())\nvocab[:20]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:23.632597Z","iopub.execute_input":"2022-08-14T04:26:23.632950Z","iopub.status.idle":"2022-08-14T04:26:23.665054Z","shell.execute_reply.started":"2022-08-14T04:26:23.632914Z","shell.execute_reply":"2022-08-14T04:26:23.664183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example = 'This is a bad test'\nvectorization_layer(example).numpy()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:23.666692Z","iopub.execute_input":"2022-08-14T04:26:23.667049Z","iopub.status.idle":"2022-08-14T04:26:23.685101Z","shell.execute_reply.started":"2022-08-14T04:26:23.667014Z","shell.execute_reply":"2022-08-14T04:26:23.684251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.Sequential([\n    vectorization_layer,\n    tf.keras.layers.Embedding(\n        input_dim=len(vectorization_layer.get_vocabulary()),\n        output_dim=64,\n        mask_zero=True\n    ),\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(64, return_sequences=True)),\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(32)),\n    tf.keras.layers.Dense(64, activation='relu'),\n    tf.keras.layers.Dropout(.5),\n    tf.keras.layers.Dense(1)\n])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:23.687631Z","iopub.execute_input":"2022-08-14T04:26:23.687945Z","iopub.status.idle":"2022-08-14T04:26:26.863690Z","shell.execute_reply.started":"2022-08-14T04:26:23.687913Z","shell.execute_reply":"2022-08-14T04:26:26.862551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n             optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n             metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:26.865172Z","iopub.execute_input":"2022-08-14T04:26:26.866139Z","iopub.status.idle":"2022-08-14T04:26:26.876234Z","shell.execute_reply.started":"2022-08-14T04:26:26.866098Z","shell.execute_reply":"2022-08-14T04:26:26.875274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_ds,\n                   epochs=10,\n                   validation_data=val_ds\n                   )","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:26:26.877514Z","iopub.execute_input":"2022-08-14T04:26:26.877854Z","iopub.status.idle":"2022-08-14T04:28:05.507850Z","shell.execute_reply.started":"2022-08-14T04:26:26.877818Z","shell.execute_reply":"2022-08-14T04:28:05.506906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist_dict = history.history\nacc, val_acc = hist_dict['accuracy'], hist_dict['val_accuracy']\nloss, val_loss = hist_dict['loss'], hist_dict['val_loss']\n\nplt.figure(figsize=(10,8))\nax1 = plt.subplot(1, 2, 1)\nax1.plot(acc, 'bo', label='Training accuracy')\nax1.plot(val_acc, 'b', label='Validation accuracy')\nax1.legend()\n\nax2 = plt.subplot(1, 2, 2)\nax2.plot(loss, 'bo', label='Training loss')\nax2.plot(val_loss, 'b', label='Validation loss')\nax2.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T04:35:26.172798Z","iopub.execute_input":"2022-08-14T04:35:26.173163Z","iopub.status.idle":"2022-08-14T04:35:26.487236Z","shell.execute_reply.started":"2022-08-14T04:35:26.173130Z","shell.execute_reply":"2022-08-14T04:35:26.486269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('../input/nlp-getting-started/test.csv')\ntest_df = test_df['text']\n\ny_preds = model.predict(test_df).round().reshape(test_df.shape[0])\npd.Series(y_preds).value_counts().sort_index().plot(kind='barh')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T05:59:48.581045Z","iopub.execute_input":"2022-08-14T05:59:48.581435Z","iopub.status.idle":"2022-08-14T05:59:49.603292Z","shell.execute_reply.started":"2022-08-14T05:59:48.581382Z","shell.execute_reply":"2022-08-14T05:59:49.602343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsubmission = pd.read_csv(\"/kaggle/input/nlp-getting-started/sample_submission.csv\")\nsubmission['target'] = [1 if p >= .0  else 0 for p in y_preds]\nsubmission.to_csv('submission.csv', index=False)\nsubmission.describe().style.background_gradient(cmap='coolwarm')","metadata":{"execution":{"iopub.status.busy":"2022-08-14T06:01:36.809872Z","iopub.execute_input":"2022-08-14T06:01:36.810248Z","iopub.status.idle":"2022-08-14T06:01:36.844965Z","shell.execute_reply.started":"2022-08-14T06:01:36.810215Z","shell.execute_reply":"2022-08-14T06:01:36.844104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}