{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-26T07:25:08.646515Z","iopub.execute_input":"2022-07-26T07:25:08.646928Z","iopub.status.idle":"2022-07-26T07:25:08.655962Z","shell.execute_reply.started":"2022-07-26T07:25:08.646866Z","shell.execute_reply":"2022-07-26T07:25:08.654720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#データの読み込み\ntrain_df=pd.read_csv('../input/nlp-getting-started/train.csv')\ntest_df=pd.read_csv('../input/nlp-getting-started/test.csv')\nsub_df=pd.read_csv('../input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:25:08.659894Z","iopub.execute_input":"2022-07-26T07:25:08.660604Z","iopub.status.idle":"2022-07-26T07:25:08.729740Z","shell.execute_reply.started":"2022-07-26T07:25:08.660567Z","shell.execute_reply":"2022-07-26T07:25:08.728731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#trainデータの中身を見てみる\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:25:08.731698Z","iopub.execute_input":"2022-07-26T07:25:08.732084Z","iopub.status.idle":"2022-07-26T07:25:08.759726Z","shell.execute_reply.started":"2022-07-26T07:25:08.732048Z","shell.execute_reply":"2022-07-26T07:25:08.758893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#テキスト前処理用のライブラリをインポート\n!pip install text_hammer\nimport text_hammer as th","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:25:08.761149Z","iopub.execute_input":"2022-07-26T07:25:08.761501Z","iopub.status.idle":"2022-07-26T07:25:31.137088Z","shell.execute_reply.started":"2022-07-26T07:25:08.761467Z","shell.execute_reply":"2022-07-26T07:25:31.135942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#進捗バーを表示させる\nfrom tqdm._tqdm_notebook import tqdm_notebook\ntqdm_notebook.pandas()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:25:31.143779Z","iopub.execute_input":"2022-07-26T07:25:31.146303Z","iopub.status.idle":"2022-07-26T07:25:31.153965Z","shell.execute_reply.started":"2022-07-26T07:25:31.146261Z","shell.execute_reply":"2022-07-26T07:25:31.152947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#データクレンジング\ndef text_preprocessing(df,col_name):\n    column = col_name\n    df[column] = df[column].progress_apply(lambda x:str(x).lower())\n    df[column] = df[column].progress_apply(lambda x: th.cont_exp(x))\n    df[column] = df[column].progress_apply(lambda x: th.remove_emails(x))\n    df[column] = df[column].progress_apply(lambda x: th.remove_html_tags(x))\n    df[column] = df[column].progress_apply(lambda x: th.remove_special_chars(x))\n    df[column] = df[column].progress_apply(lambda x: th.remove_accented_chars(x))\n    df[column] = df[column].progress_apply(lambda x: th.make_base(x))\n    \n    return(df)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:25:31.158786Z","iopub.execute_input":"2022-07-26T07:25:31.161302Z","iopub.status.idle":"2022-07-26T07:25:31.175359Z","shell.execute_reply.started":"2022-07-26T07:25:31.161264Z","shell.execute_reply":"2022-07-26T07:25:31.174071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cleaned_data = text_preprocessing(train_df,'text')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:25:31.177240Z","iopub.execute_input":"2022-07-26T07:25:31.178046Z","iopub.status.idle":"2022-07-26T07:27:02.898471Z","shell.execute_reply.started":"2022-07-26T07:25:31.178005Z","shell.execute_reply":"2022-07-26T07:27:02.897247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#クリーンデータの中身を見てみる\ntrain_cleaned_data","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:27:02.899966Z","iopub.execute_input":"2022-07-26T07:27:02.900622Z","iopub.status.idle":"2022-07-26T07:27:02.919311Z","shell.execute_reply.started":"2022-07-26T07:27:02.900581Z","shell.execute_reply":"2022-07-26T07:27:02.918298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#コピーしたクリーンデータをtrainデータとして使用\ntrain_df = train_cleaned_data.copy()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:27:02.920635Z","iopub.execute_input":"2022-07-26T07:27:02.921432Z","iopub.status.idle":"2022-07-26T07:27:02.926449Z","shell.execute_reply.started":"2022-07-26T07:27:02.921396Z","shell.execute_reply":"2022-07-26T07:27:02.925358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#いらない列を削除\ntrain_df = train_df.drop(columns=['id', 'keyword', 'location'], axis=1)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:27:02.928010Z","iopub.execute_input":"2022-07-26T07:27:02.929084Z","iopub.status.idle":"2022-07-26T07:27:02.942204Z","shell.execute_reply.started":"2022-07-26T07:27:02.929046Z","shell.execute_reply":"2022-07-26T07:27:02.940918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#正解ラベル0,1の数を数える\ntrain_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:27:02.947991Z","iopub.execute_input":"2022-07-26T07:27:02.948511Z","iopub.status.idle":"2022-07-26T07:27:02.959658Z","shell.execute_reply.started":"2022-07-26T07:27:02.948481Z","shell.execute_reply":"2022-07-26T07:27:02.958524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#transformersライブラリのインストール\nimport transformers\nfrom transformers import AutoTokenizer,TFBertModel\n\n#エンコーダの呼び出し\ntokenizer=AutoTokenizer.from_pretrained('bert-large-uncased')\n#事前学習モデルの呼び出し\nbert=TFBertModel.from_pretrained('bert-large-uncased')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:27:02.961386Z","iopub.execute_input":"2022-07-26T07:27:02.962253Z","iopub.status.idle":"2022-07-26T07:28:48.660227Z","shell.execute_reply.started":"2022-07-26T07:27:02.962217Z","shell.execute_reply":"2022-07-26T07:28:48.659246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#trainデータ内の最大文字数を表示\nprint('max length of words is ',max([len(x.split()) for x in train_df.text]))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:48.661966Z","iopub.execute_input":"2022-07-26T07:28:48.662669Z","iopub.status.idle":"2022-07-26T07:28:48.677521Z","shell.execute_reply.started":"2022-07-26T07:28:48.662629Z","shell.execute_reply":"2022-07-26T07:28:48.676547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#最大文字数を指定\nmax_length=40","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:48.679225Z","iopub.execute_input":"2022-07-26T07:28:48.679575Z","iopub.status.idle":"2022-07-26T07:28:48.687064Z","shell.execute_reply.started":"2022-07-26T07:28:48.679540Z","shell.execute_reply":"2022-07-26T07:28:48.686168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#trainデータをBERTで使えるように整形\nx_train=tokenizer( #単語分割・IDへ変換 \n        \n    text=train_df['text'].tolist(),\n    add_special_tokens=True, #文章の最後に[SEP],始めに[CLS]という単語を追加する\n    max_length=max_length,\n    truncation=True, #固定長を超える長さは切り捨て\n    padding=True, #固定長に満たない場合は埋める\n    return_tensors='tf', #TensorFlowでテンソルを返す\n    return_token_type_ids=True, #文を判別するバイナリマスクを返す\n    return_attention_mask=True, #埋め込みを判別するバイナリマスクを返す\n    verbose=True #ログ出力\n    \n    )","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:48.688469Z","iopub.execute_input":"2022-07-26T07:28:48.689204Z","iopub.status.idle":"2022-07-26T07:28:49.384300Z","shell.execute_reply.started":"2022-07-26T07:28:48.689124Z","shell.execute_reply":"2022-07-26T07:28:49.383274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#正解ラベルの値を取得\ny_train=train_df['target'].values\ny_train","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:49.385611Z","iopub.execute_input":"2022-07-26T07:28:49.387754Z","iopub.status.idle":"2022-07-26T07:28:49.395826Z","shell.execute_reply.started":"2022-07-26T07:28:49.387714Z","shell.execute_reply":"2022-07-26T07:28:49.394657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#機械学習ライブラリtensorflowをインポート\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Input,Dense\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.initializers import TruncatedNormal\nfrom tensorflow.keras.losses import CategoricalCrossentropy,BinaryCrossentropy\nfrom tensorflow.keras.metrics import CategoricalAccuracy,BinaryAccuracy\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.utils import plot_model","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:49.397511Z","iopub.execute_input":"2022-07-26T07:28:49.398251Z","iopub.status.idle":"2022-07-26T07:28:49.408813Z","shell.execute_reply.started":"2022-07-26T07:28:49.398187Z","shell.execute_reply":"2022-07-26T07:28:49.407813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#学習データ構造の入力シーケンス\n# BERTトークナイザーから符号化されたトークンID\ninput_ids=Input(shape=(max_length,), dtype=tf.int32, name='input_ids')\n#どのトークンにアテンションすべきかをモデルに示す\ninput_mask=Input(shape=(max_length,),dtype=tf.int32 ,name='attention_mask')\n#モデル内の異なるシーケンスを識別するバイナリマスク\ntoken_ids=Input(shape=(max_length,), dtype=tf.int32, name='token_type_ids')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:49.410499Z","iopub.execute_input":"2022-07-26T07:28:49.411305Z","iopub.status.idle":"2022-07-26T07:28:49.423647Z","shell.execute_reply.started":"2022-07-26T07:28:49.411271Z","shell.execute_reply":"2022-07-26T07:28:49.422635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding=bert(input_ids,attention_mask=input_mask,token_type_ids=token_ids)[1] #呼び出した事前学習モデルにトークン列を入力\nout=tf.keras.layers.Dropout(0.1)(embedding)\n\nout=Dense(120,activation='relu')(out)\nout=tf.keras.layers.Dropout(0.11)(out)\nout=Dense(31,activation='relu')(out)\n\ny=Dense(1,activation='relu')(out)\n\nmodel=tf.keras.Model(inputs=[input_ids,input_mask,token_ids],outputs=y)\nmodel.layers[2].trainable=True","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:49.425000Z","iopub.execute_input":"2022-07-26T07:28:49.425497Z","iopub.status.idle":"2022-07-26T07:28:57.960174Z","shell.execute_reply.started":"2022-07-26T07:28:49.425460Z","shell.execute_reply":"2022-07-26T07:28:57.959165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:57.961435Z","iopub.execute_input":"2022-07-26T07:28:57.962126Z","iopub.status.idle":"2022-07-26T07:28:57.996687Z","shell.execute_reply.started":"2022-07-26T07:28:57.962085Z","shell.execute_reply":"2022-07-26T07:28:57.995543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer=Adam(\n \n    learning_rate=5e-6,\n    epsilon=1e-8,\n    decay=0.01,\n    clipnorm=1.0\n\n)\n\nloss=BinaryCrossentropy(from_logits=True)\nmetrics=BinaryAccuracy('accuracy')\nmodel.compile(optimizer=optimizer,loss=loss,metrics=metrics) #トレーニング前にモデルをコンパイル","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:57.998701Z","iopub.execute_input":"2022-07-26T07:28:57.999538Z","iopub.status.idle":"2022-07-26T07:28:58.031160Z","shell.execute_reply.started":"2022-07-26T07:28:57.999496Z","shell.execute_reply":"2022-07-26T07:28:58.030102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(model=model,show_shapes=True,show_dtype=True,expand_nested=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:58.032397Z","iopub.execute_input":"2022-07-26T07:28:58.032750Z","iopub.status.idle":"2022-07-26T07:28:59.205873Z","shell.execute_reply.started":"2022-07-26T07:28:58.032713Z","shell.execute_reply":"2022-07-26T07:28:59.204858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.config.experimental.list_physical_devices('GPU')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:59.207570Z","iopub.execute_input":"2022-07-26T07:28:59.207994Z","iopub.status.idle":"2022-07-26T07:28:59.219031Z","shell.execute_reply.started":"2022-07-26T07:28:59.207950Z","shell.execute_reply":"2022-07-26T07:28:59.218012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history=model.fit(\n\n    x={'input_ids':x_train['input_ids'],'attention_mask':x_train['attention_mask'],'token_type_ids':x_train['token_type_ids']},\n    y=y_train,\n    validation_split=0.20,\n    epochs=3,\n    batch_size=45\n\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:28:59.220395Z","iopub.execute_input":"2022-07-26T07:28:59.220825Z","iopub.status.idle":"2022-07-26T07:36:06.615034Z","shell.execute_reply.started":"2022-07-26T07:28:59.220789Z","shell.execute_reply":"2022-07-26T07:36:06.613833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"テストデータで試す","metadata":{}},{"cell_type":"code","source":"#trainデータと同じように整形\ntest_df=test_df.drop(columns=['id','keyword','location'],axis=1)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:36:06.617151Z","iopub.execute_input":"2022-07-26T07:36:06.617546Z","iopub.status.idle":"2022-07-26T07:36:06.633469Z","shell.execute_reply.started":"2022-07-26T07:36:06.617507Z","shell.execute_reply":"2022-07-26T07:36:06.632317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#trainデータと同じようにクレンジング\ntest_df=text_preprocessing(test_df,'text')\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:36:06.635145Z","iopub.execute_input":"2022-07-26T07:36:06.635914Z","iopub.status.idle":"2022-07-26T07:36:43.790854Z","shell.execute_reply.started":"2022-07-26T07:36:06.635844Z","shell.execute_reply":"2022-07-26T07:36:43.789808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#testデータをBERTで使えるように整形\nx_test=tokenizer(\n\n    text=test_df.text.tolist(),\n    add_special_tokens=True,\n    max_length=max_length,\n    truncation=True,\n    padding=True,\n    return_tensors='tf',\n    return_token_type_ids=True,\n    return_attention_mask=True,\n    verbose=True\n\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:36:43.792390Z","iopub.execute_input":"2022-07-26T07:36:43.792833Z","iopub.status.idle":"2022-07-26T07:36:44.075160Z","shell.execute_reply.started":"2022-07-26T07:36:43.792794Z","shell.execute_reply":"2022-07-26T07:36:44.074181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#学習済みモデルを使った予測（推論）\npredicted=model.predict({'input_ids':x_test['input_ids'],'attention_mask':x_test['attention_mask'],'token_type_ids':x_test['token_type_ids']})\npredicted\npredicted.shape #配列の各次元ごとの要素数","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:36:44.076855Z","iopub.execute_input":"2022-07-26T07:36:44.077240Z","iopub.status.idle":"2022-07-26T07:37:08.495790Z","shell.execute_reply.started":"2022-07-26T07:36:44.077201Z","shell.execute_reply":"2022-07-26T07:37:08.494785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#正解ラベルの予測\ny_predicted=np.where(predicted>0.5,1,0) #予測値が0.5より大きいなら1,それ以外は0\ny_predicted","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:37:08.501801Z","iopub.execute_input":"2022-07-26T07:37:08.503430Z","iopub.status.idle":"2022-07-26T07:37:08.511031Z","shell.execute_reply.started":"2022-07-26T07:37:08.503388Z","shell.execute_reply":"2022-07-26T07:37:08.509859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#提出用ファイルのサンプルの中身を見てみる\nsub_df","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:37:08.512816Z","iopub.execute_input":"2022-07-26T07:37:08.513618Z","iopub.status.idle":"2022-07-26T07:37:08.530438Z","shell.execute_reply.started":"2022-07-26T07:37:08.513581Z","shell.execute_reply":"2022-07-26T07:37:08.529426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_predicted=y_predicted.reshape(-1,) #-1を指定すると具体値を指定しなくても自動的にもう片方の数値から適当な数値を決めてくれる\ny_predicted","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:37:08.531942Z","iopub.execute_input":"2022-07-26T07:37:08.532589Z","iopub.status.idle":"2022-07-26T07:37:08.539998Z","shell.execute_reply.started":"2022-07-26T07:37:08.532554Z","shell.execute_reply":"2022-07-26T07:37:08.538872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#提出用ファイルを作成\nsubmission=pd.DataFrame({'id':sub_df['id'],'target':y_predicted})\nsubmission.tail()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:37:08.541656Z","iopub.execute_input":"2022-07-26T07:37:08.542642Z","iopub.status.idle":"2022-07-26T07:37:08.553225Z","shell.execute_reply.started":"2022-07-26T07:37:08.542492Z","shell.execute_reply":"2022-07-26T07:37:08.552071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#提出用ファイルの正解ラベルの数を数える\nsubmission.target.value_counts(normalize=True) #相対的な割合で表示","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:37:08.554776Z","iopub.execute_input":"2022-07-26T07:37:08.555453Z","iopub.status.idle":"2022-07-26T07:37:08.565706Z","shell.execute_reply.started":"2022-07-26T07:37:08.555413Z","shell.execute_reply":"2022-07-26T07:37:08.564566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#csvファイルに変換\nsubmission.to_csv('datascience6_nlp.csv',index=False)\nprint('Submission succesful!')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T07:37:08.567225Z","iopub.execute_input":"2022-07-26T07:37:08.567868Z","iopub.status.idle":"2022-07-26T07:37:08.581854Z","shell.execute_reply.started":"2022-07-26T07:37:08.567830Z","shell.execute_reply":"2022-07-26T07:37:08.580877Z"},"trusted":true},"execution_count":null,"outputs":[]}]}