{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7526248,"sourceType":"datasetVersion","datasetId":4308295},{"sourceId":6127,"sourceType":"modelInstanceVersion","modelInstanceId":4598}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#パッケージをインストール\n!pip install -q /kaggle/input/kerasv3-lib-ds/keras_cv-0.8.2-py3-none-any.whl --no-deps\n!pip install -q /kaggle/input/kerasv3-lib-ds/tensorflow-2.15.0.post1-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n!pip install -q /kaggle/input/kerasv3-lib-ds/keras-3.0.4-py3-none-any.whl --no-deps\n       #「--no-deps」オプションを指定することで、追加のライブラリがインストールされないようにすることができる\n\nimport tensorflow as tf\nimport keras\nimport keras_cv\n\n\nprint(\"TensorFlow:\", tf.__version__)\nprint(\"Keras:\", keras.__version__)\nprint(\"KerasCV:\", keras_cv.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:20:58.888242Z","iopub.execute_input":"2024-04-03T01:20:58.888956Z","iopub.status.idle":"2024-04-03T01:22:15.674730Z","shell.execute_reply.started":"2024-04-03T01:20:58.888923Z","shell.execute_reply":"2024-04-03T01:22:15.673723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#kerasの乱数固定\nkeras.utils.set_random_seed(42)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:22:25.940111Z","iopub.execute_input":"2024-04-03T01:22:25.941167Z","iopub.status.idle":"2024-04-03T01:22:25.945700Z","shell.execute_reply.started":"2024-04-03T01:22:25.941129Z","shell.execute_reply":"2024-04-03T01:22:25.944820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\n#CSVファイルを読み込みデータフレームにデータを格納\ndf=pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\nprint(df)\ndf['class_label'] = df.expert_consensus.map(dict((j,i) for i,j in enumerate(['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other'])))\n                    # 注釈者のラベルを数値に変換したものをデータフレームに追加","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:22:28.069896Z","iopub.execute_input":"2024-04-03T01:22:28.070653Z","iopub.status.idle":"2024-04-03T01:22:28.473142Z","shell.execute_reply.started":"2024-04-03T01:22:28.070617Z","shell.execute_reply":"2024-04-03T01:22:28.472019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.makedirs(\"/tmp/dataset/hms-hbac/train_spectrograms\", exist_ok=True)\nos.makedirs(\"/tmp/dataset/hms-hbac/test_spectrograms\", exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:22:32.798151Z","iopub.execute_input":"2024-04-03T01:22:32.798894Z","iopub.status.idle":"2024-04-03T01:22:32.808872Z","shell.execute_reply.started":"2024-04-03T01:22:32.798850Z","shell.execute_reply":"2024-04-03T01:22:32.807889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib\nfrom tqdm.notebook import tqdm\n\nspec_path_ex = f\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet\"\nspec_ex = pd.read_parquet(spec_path_ex)\nprint(spec_ex.isnull().any()) #データにNaNが含まれていることを確認\n\n\n\n##### spectrogramデータをバイナリファイルに保存\ndef process_spec(spec_id, split=\"train\"):\n    spec_path = f\"/kaggle/input/hms-harmful-brain-activity-classification/{split}_spectrograms/{spec_id}.parquet\"\n    spec = pd.read_parquet(spec_path)\n    spec = spec.fillna(0).values[:, 1:].T #NaNを0で埋めて、要素を転置し、列を周波数、行をTimeとする\n    spec = spec.astype(\"float32\") #データ型をfloat64→float32に変換\n    np.save(f\"/tmp/dataset/hms-hbac/{split}_spectrograms/{spec_id}.npy\", spec) #バイナリファイルに保存\n\nspec_ids = df[\"spectrogram_id\"].unique()\n    \n_= joblib.Parallel(n_jobs=-1)(joblib.delayed(process_spec)(spec_id)\n                                   for spec_id in tqdm(spec_ids))  #n_jobs=-1として、joblibで全てのCPUを使用し並列処理 \n#(backend=\"loky\"削除)\ndf['spec2_path'] = f'/tmp/dataset/hms-hbac/train_spectrograms/'+df['spectrogram_id'].astype(str)+'.npy'\n                 # 保存したバイナリファイルのpathをデータフレームに追加","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:22:35.623926Z","iopub.execute_input":"2024-04-03T01:22:35.625390Z","iopub.status.idle":"2024-04-03T01:26:28.974276Z","shell.execute_reply.started":"2024-04-03T01:22:35.625347Z","shell.execute_reply":"2024-04-03T01:26:28.972533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(50):\n    print(np.load(df.loc[i,'spec2_path']).shape) #先頭50個のspectrogramsデータのshapeは、400行、300列以上\n\nprint(\"----------------\")\n\nfor i in range(len(df)):\n    if np.load(df.loc[i,'spec2_path']).shape[0] !=400:\n        print(np.load(df.loc[i,'spec2_path']).shape[0]) #全てのspectrogramsデータの行数は400で揃っている\nprint(\"end\")\n\nprint(\"----------------\")\n\nfor i in range(len(df)):\n    if np.load(df.loc[i,'spec2_path']).shape[1] <300:\n        print(np.load(df.loc[i,'spec2_path']).shape[1]) #全てのspectrogramsデータの列数は300以上である\nprint(\"end\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n#/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1002360431.parquet\ndata=np.load(\"/tmp/dataset/hms-hbac/train_spectrograms/1002360431.npy\")\nprint(data.max())\n#a=plt.hist(data)\n#print(a)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.model_selection import StratifiedGroupKFold\n\n##### train, test, validationデータの分割  \nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=42) \ndf[\"fold\"] = -1\ndf.reset_index(drop=True, inplace=True)\nfor fold, (train_idx, valid_idx) in enumerate(\n    sgkf.split(df, y=df[\"expert_consensus\"], groups=df[\"patient_id\"])\n):\n    #データ漏洩を防ぐため、同じ患者のデータは同じ側のデータにし、更に各フォールドでクラスラベルの分布が均一にする\n    df.loc[valid_idx, \"fold\"] = fold\n        #validationデータにfold番号を追加\ndf_dis=df.groupby([\"fold\", \"expert_consensus\"])[[\"eeg_id\"]].count().T\nprint(df_dis)\n\"\"\"for i in range(5):\n    ax=plt.subplot(5,1,i+1)\n    ax.bar(df_dis[i].T.reset_index()[\"expert_consensus\"],df_dis[i].T.reset_index()[\"eeg_id\"])\nfor i in range(5):\n    print(f\"Fold:{i}\")\n    for j in range(6):\n        print(df_dis[i].T.reset_index().loc[j,\"eeg_id\"]/df_dis[i].T.reset_index()[\"eeg_id\"].sum())\n#各foldでのラベルの分布を確認\"\"\"\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:29:09.330087Z","iopub.execute_input":"2024-04-03T01:29:09.330557Z","iopub.status.idle":"2024-04-03T01:29:12.561717Z","shell.execute_reply.started":"2024-04-03T01:29:09.330521Z","shell.execute_reply":"2024-04-03T01:29:12.560468Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sample from full data\nsample_df = df.groupby(\"spectrogram_id\").head(1).reset_index(drop=True)\n#sample_df = df.groupby(\"spectrogram_id\").reset_index(drop=True)\ntrain_df = sample_df[sample_df.fold != 0]\nvalid_df = sample_df[sample_df.fold == 0]\nprint(f\"# Num Train: {len(train_df)} | Num Valid: {len(valid_df)}\")\n\ntrain_paths = train_df.spec2_path.values\ntrain_labels = train_df.class_label.values\n\nvalid_paths = valid_df.spec2_path.values\nvalid_labels = valid_df.class_label.values","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:29:17.759505Z","iopub.execute_input":"2024-04-03T01:29:17.759903Z","iopub.status.idle":"2024-04-03T01:29:17.787005Z","shell.execute_reply.started":"2024-04-03T01:29:17.759875Z","shell.execute_reply":"2024-04-03T01:29:17.785745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### 一つのpathとlabelを学習データに変換する関数\ndef build_decoder(with_labels=True):\n    ### pathを学習データに変換する関数\n    def decode_signal(path):\n        file_bytes = tf.io.read_file(path)\n        sig = tf.io.decode_raw(file_bytes, tf.float32)\n        sig = sig[1024//32:] #ヘッダーのダグを取り除く\n        sig = tf.reshape(sig, [400, -1])\n        sig = sig[:, 0:300]  #shapeを(400,300)に整える\n        \n        sig = tf.clip_by_value(sig, tf.math.exp(-4.0), tf.math.exp(8.0)) # 0を避けるためにクリッピング\n        sig = tf.math.log(sig)   #対数を取る\n        \n        sig -= tf.math.reduce_mean(sig)\n        sig /= tf.math.reduce_std(sig) + 1e-6   # 標準化\n        \n        sig = tf.tile(sig[..., None], [1, 1, 3])   # ImageNetの重みを使うため、モノラルチャンネルを3チャネルに変更\n        return sig\n    \n    ### labelを学習データに変換する関数\n    def decode_label(label):\n        label = tf.one_hot(label, 6)\n        label = tf.cast(label, tf.float32)\n        label = tf.reshape(label, [6])#shape=(1,6)→shape=(6,)\n        return label\n    \n    def decode(path,label):\n        sig = decode_signal(path)\n        label = decode_label(label)\n        return (sig, label)\n    return decode if with_labels else decode_signal\n\n##### path、labelからdatasetを作成する関数\ndef build_dataset(paths,labels=None):\n    decode_fn = build_decoder(labels is not None)\n    slices = (paths) if labels is None else (paths,labels)\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=tf.data.experimental.AUTOTUNE) #GPUの処理とCPUの処理の配分を動的に設定\n    ds = ds.cache() #　 キャッシュをファイルに保存することで、メモリに収まらないデータもキャッシュする。\n                    #　 計算負荷の高い作業の後に挿入することで、メモリの使用量は増えるが、不要な演算を回避する。\n    ds = ds.repeat()\n    ds = ds.shuffle(1024,seed=42)\n    opt = tf.data.Options()\n    opt.experimental_deterministic = False\n    ds =ds.with_options(opt)  # 計算の並列処理を動的に調整\n    ds = ds.batch(64, drop_remainder=False) \n    ds = ds.prefetch(tf.data.experimental.AUTOTUNE)  #　トレーニングステップsを実行する間、ステップs+1のデータを読み取ることで、データの抽出にかかる時間を減少させる\n    return ds\n\ntrain_ds = build_dataset(train_paths,train_labels)\n\nvalid_ds=build_dataset(valid_paths,valid_labels)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:29:31.075665Z","iopub.execute_input":"2024-04-03T01:29:31.076100Z","iopub.status.idle":"2024-04-03T01:29:31.598002Z","shell.execute_reply.started":"2024-04-03T01:29:31.076069Z","shell.execute_reply":"2024-04-03T01:29:31.597127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs, tars = next(iter(train_ds))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#####   データセットの可視化\n\nimgs, tars = next(iter(train_ds)) # 1バッチに含まれる64個のデータ\nprint(tars)\nlabel2name = dict(enumerate(['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other']))\nprint(label2name)\nplt.figure(figsize=(4*4, 2*5))\nfor i in range(8):\n    plt.subplot(2, 4, i + 1)\n    img = imgs[i].numpy()[...,0]  # 3チャネルをモノラルチャンネルに変更\n    img -= img.min()\n    img /= img.max() + 1e-4\n    tar = label2name[np.argmax(tars[i].numpy())]\n    plt.imshow(img)\n    plt.title(f\"Target: {tar}\")\n    plt.axis('off')\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:29:35.298281Z","iopub.execute_input":"2024-04-03T01:29:35.298902Z","iopub.status.idle":"2024-04-03T01:29:41.793581Z","shell.execute_reply.started":"2024-04-03T01:29:35.298872Z","shell.execute_reply":"2024-04-03T01:29:41.791287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#####  モデルの作成\nmodel = keras_cv.models.ImageClassifier.from_preset(\n    \"efficientnetv2_b2_imagenet\", num_classes=6)   #　EfficientNetV2 B2:画像分類タスクに特化した軽量なニューラルネットワークモデル　　　\n    \n    \nmodel.compile(optimizer=keras.optimizers.Adam(learning_rate=1e-4),\n              loss=keras.losses.KLDivergence())   #評価指標はKLダイバージェンス\n\nmodel.summary()  # Model Sumamry","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:31:16.464489Z","iopub.execute_input":"2024-04-03T01:31:16.464912Z","iopub.status.idle":"2024-04-03T01:31:28.154621Z","shell.execute_reply.started":"2024-04-03T01:31:16.464882Z","shell.execute_reply":"2024-04-03T01:31:28.153433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n\n##### 学習率スケジュール：\n\n## 学習過程における学習率の変化を制御する、学習率スケジュールのコールバックの関数\ndef get_lr_callback(batch_size=8, epochs=10):\n    lr_start, lr_max, lr_min = 5e-5, 6e-6 * batch_size, 1e-5\n    lr_ramp_ep, lr_sus_ep, lr_decay = 3, 0, 0.75\n    #lr_start: 学習開始時の学習率 (デフォルト: 5e-5)\n    #lr_max: 学習率の最大値 (デフォルト: バッチサイズ * 6e-6)\n    #lr_min: 学習率の最小値 (デフォルト: 1e-5)\n    #lr_ramp_ep: 学習率を徐々に上げる期間のエポック数 (デフォルト: 3)\n    #lr_sus_ep: 学習率を最大値に維持する期間のエポック数 (デフォルト: 0)\n    #lr_decay: 学習率を減衰させる際の減衰率 (デフォルト: 0.75)\n\n    def lrfn(epoch):  # Learning rate update function\n        if epoch < lr_ramp_ep: lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n        elif epoch < lr_ramp_ep + lr_sus_ep: lr = lr_max\n        else:\n            decay_total_epochs, decay_epoch_index = epochs - lr_ramp_ep - lr_sus_ep + 3, epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            lr = (lr_max - lr_min) * 0.5 * (1 + math.cos(phase)) + lr_min   #　コサイン関数を使って周期的に学習率を変化させる。lrはepoch3以上は徐々に減少。\n        return lr\n\n\n    plt.figure(figsize=(10, 5))\n    plt.plot(np.arange(epochs), [lrfn(epoch) for epoch in np.arange(epochs)], marker='o')\n    plt.xlabel('epoch'); plt.ylabel('lr')\n    plt.title('LR Scheduler')\n    plt.show()\n\n    return keras.callbacks.LearningRateScheduler(lrfn, verbose=False)  \n           # keras.callbacks.LearningRateSchedulerコールバックは、学習のたびに lrfn関数を呼び出して、更新された学習率を適用。\n\nlr_cb = get_lr_callback(64)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:32:07.745218Z","iopub.execute_input":"2024-04-03T01:32:07.745718Z","iopub.status.idle":"2024-04-03T01:32:08.076265Z","shell.execute_reply.started":"2024-04-03T01:32:07.745684Z","shell.execute_reply":"2024-04-03T01:32:08.075098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##### モデルのチェックポイントを保存\n\nckpt_cb = keras.callbacks.ModelCheckpoint(\"best_model.keras\",\n                                         monitor='val_loss',\n                                         save_best_only=True,\n                                         save_weights_only=False,  # モデルと重みの両方を保存\n                                         mode='min')  # 最適化方向は最小化 ","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:32:10.734338Z","iopub.execute_input":"2024-04-03T01:32:10.734773Z","iopub.status.idle":"2024-04-03T01:32:10.740922Z","shell.execute_reply.started":"2024-04-03T01:32:10.734741Z","shell.execute_reply":"2024-04-03T01:32:10.739528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#####  モデルの学習\n\nhistory = model.fit(\n    train_ds, \n    epochs=13,\n    callbacks=[lr_cb, ckpt_cb], \n    steps_per_epoch=len(train_df)//64,\n    validation_data=valid_ds, \n    verbose=1\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#####  最良のモデルを読み込み\n\nmodel.load_weights(\"best_model.keras\")\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#####  学習したモデルを使って予測\n\ntest_df=pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")\nprint(test_df)\n\ntest_spec_ids = test_df[\"spectrogram_id\"].unique()\n\n_ = joblib.Parallel(n_jobs=-1)(joblib.delayed(process_spec)(spec_id,\"test\")\n    for spec_id in tqdm(test_spec_ids))\n\n\ntest_df['spec2_path'] = f'/tmp/dataset/hms-hbac/test_spectrograms/'+test_df['spectrogram_id'].astype(str)+'.npy'\n\ntest_paths = test_df.spec2_path.values\ntest_ds = build_dataset(test_paths)\npreds = model.predict(test_ds)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#　予測した値をsubmit\n\nsub_df = pd.read_csv(f'/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')\nsub_df = sub_df[[\"eeg_id\"]].copy()\ntarget_cols = [x.lower()+'_vote' for x in ['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other']]\nsub_df[target_cols] = preds.tolist()\nsub_df.to_csv(\"submission.csv\", index=False)\nprint(sub_df)","metadata":{},"execution_count":null,"outputs":[]}]}