{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-22T05:32:25.652268Z","iopub.execute_input":"2022-07-22T05:32:25.652724Z","iopub.status.idle":"2022-07-22T05:32:25.692418Z","shell.execute_reply.started":"2022-07-22T05:32:25.652617Z","shell.execute_reply":"2022-07-22T05:32:25.690902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/nlp-getting-started/train.csv')\ntest = pd.read_csv('../input/nlp-getting-started/test.csv')\nsub = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:32:25.695039Z","iopub.execute_input":"2022-07-22T05:32:25.695491Z","iopub.status.idle":"2022-07-22T05:32:25.850108Z","shell.execute_reply.started":"2022-07-22T05:32:25.695449Z","shell.execute_reply":"2022-07-22T05:32:25.848673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop(['id','keyword','location'],axis=1)\n#train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:32:25.852540Z","iopub.execute_input":"2022-07-22T05:32:25.853473Z","iopub.status.idle":"2022-07-22T05:32:25.870586Z","shell.execute_reply.started":"2022-07-22T05:32:25.853413Z","shell.execute_reply":"2022-07-22T05:32:25.869235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\n\ndef cleans(input):\n    output = []\n    \n    for i in input:\n        temp = i.lower()\n        temp = re.sub(r'https?://[\\w/:%#\\$&\\?\\(\\)~\\.=\\+\\-]+', r'', i)\n        temp = re.sub(r'#', r'', temp)\n        temp = re.sub(r'@[0-9a-zA-Z_:]*', r'', temp)\n        temp = re.sub(r'<.*?>', r'', temp)\n        emoji_pattern = re.compile(\"[\"\n                               u\"\\U0001F600-\\U0001F64F\"\n                               u\"\\U0001F300-\\U0001F5FF\"\n                               u\"\\U0001F680-\\U0001F6FF\"\n                               u\"\\U0001F1E0-\\U0001F1FF\"\n                               u\"\\U00002500-\\U00002BEF\"\n                               u\"\\U00002702-\\U000027B0\"\n                               u\"\\U00002702-\\U000027B0\"\n                               u\"\\U000024C2-\\U0001F251\"\n                               u\"\\U0001f926-\\U0001f937\"\n                               u\"\\U00010000-\\U0010ffff\"\n                               u\"\\u2640-\\u2642\"\n                               u\"\\u2600-\\u2B55\"\n                               u\"\\u200d\"\n                               u\"\\u23cf\"\n                               u\"\\u23e9\"\n                               u\"\\u231a\"\n                               u\"\\ufe0f\"\n                               u\"\\u3030\"\n                               \"]+\", flags=re.UNICODE)\n        temp = emoji_pattern.sub(r'', temp)        \n        output.append(temp)\n        \n    assert len(input) == len(output)\n    \n    return(output)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:32:25.872698Z","iopub.execute_input":"2022-07-22T05:32:25.874091Z","iopub.status.idle":"2022-07-22T05:32:25.885849Z","shell.execute_reply.started":"2022-07-22T05:32:25.874050Z","shell.execute_reply":"2022-07-22T05:32:25.884615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"text\"] = cleans(train[\"text\"])\n\ntrain = train.sample(frac=1, random_state=0, ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:32:25.889396Z","iopub.execute_input":"2022-07-22T05:32:25.890314Z","iopub.status.idle":"2022-07-22T05:32:26.113062Z","shell.execute_reply.started":"2022-07-22T05:32:25.890272Z","shell.execute_reply":"2022-07-22T05:32:26.111679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --no-deps tensorflow-text==2.6.0","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:32:26.114547Z","iopub.execute_input":"2022-07-22T05:32:26.115315Z","iopub.status.idle":"2022-07-22T05:32:30.879355Z","shell.execute_reply.started":"2022-07-22T05:32:26.115272Z","shell.execute_reply":"2022-07-22T05:32:30.877883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_text as text\n\ntext_input = tf.keras.layers.Input(shape=(), dtype=tf.string, name='text-layer')\n\npreprocess = hub.KerasLayer(\"https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3\", name='preprocessing-layer')\npreprocessed_text = preprocess(text_input)\n\nencoder = hub.KerasLayer(\"https://tfhub.dev/tensorflow/bert_en_uncased_L-12_H-768_A-12/4\", trainable=False, name='BERT_encoder')\noutputs = encoder(preprocessed_text)\n\nd_layer = outputs['pooled_output']\nd_layer = tf.keras.layers.Dropout(0.25)(d_layer)\nd_layer = tf.keras.layers.Dense(256 , activation = \"relu\")(d_layer)\nd_layer = tf.keras.layers.Dropout(0.25)(d_layer)\nd_layer = tf.keras.layers.Dense(64 , activation = \"relu\")(d_layer)\nd_layer = tf.keras.layers.Dropout(0.25)(d_layer)\nd_layer = tf.keras.layers.Dense(1, activation ='sigmoid', name =\"classifier-output\")(d_layer)\n\nmodel = tf.keras.Model(inputs=[text_input], outputs = [d_layer])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:32:30.881339Z","iopub.execute_input":"2022-07-22T05:32:30.881831Z","iopub.status.idle":"2022-07-22T05:33:10.852121Z","shell.execute_reply.started":"2022-07-22T05:32:30.881781Z","shell.execute_reply":"2022-07-22T05:33:10.850517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam', loss='binary_crossentropy', metrics='accuracy')","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:33:10.856484Z","iopub.execute_input":"2022-07-22T05:33:10.858843Z","iopub.status.idle":"2022-07-22T05:33:10.874162Z","shell.execute_reply.started":"2022-07-22T05:33:10.858810Z","shell.execute_reply":"2022-07-22T05:33:10.872923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train['text'], train['target'], batch_size=256, epochs=16, validation_split=0.1)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T05:33:10.876148Z","iopub.execute_input":"2022-07-22T05:33:10.877368Z","iopub.status.idle":"2022-07-22T06:06:43.013983Z","shell.execute_reply.started":"2022-07-22T05:33:10.877321Z","shell.execute_reply":"2022-07-22T06:06:43.012498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n#グラフの描画\n#accuracyのグラフ\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nepochs = range(1, len(acc)+1)\nplt.plot(epochs, acc, 'b', label='Training accuracy')\nplt.plot(epochs, val_acc, 'r', label='Val accuracy')\nplt.legend()\nplt.show()\n#lossのグラフ\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs = range(1, len(loss)+1 )\nplt.plot(epochs, loss, 'b', label='Training loss')\nplt.plot(epochs, val_loss, 'r', label='Val loss')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-22T06:06:43.016679Z","iopub.execute_input":"2022-07-22T06:06:43.017206Z","iopub.status.idle":"2022-07-22T06:06:43.517445Z","shell.execute_reply.started":"2022-07-22T06:06:43.017161Z","shell.execute_reply":"2022-07-22T06:06:43.516203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test.drop(['id','keyword','location'],axis=1)\ntest[\"text\"] = cleans(test[\"text\"])\n\nsub['target'] = model.predict(test['text'])\nsub['target'] = np.where(sub['target'] > 0.5, 1, 0)\n\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-22T06:06:43.519293Z","iopub.execute_input":"2022-07-22T06:06:43.519618Z","iopub.status.idle":"2022-07-22T06:07:25.398060Z","shell.execute_reply.started":"2022-07-22T06:06:43.519590Z","shell.execute_reply":"2022-07-22T06:07:25.396669Z"},"trusted":true},"execution_count":null,"outputs":[]}]}