{"metadata":{"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7526248,"sourceType":"datasetVersion","datasetId":4308295},{"sourceId":6127,"sourceType":"modelInstanceVersion","modelInstanceId":4598}],"dockerImageVersionId":30683,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"papermill":{"default_parameters":{},"duration":3846.080383,"end_time":"2024-01-14T04:20:19.064569","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-01-14T03:16:12.984186","version":"2.4.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"08983a9c6aff42578980f4f7113c3ee2":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_4411aefc021d46d0ada7b645eb53ec48","placeholder":"​","style":"IPY_MODEL_09a10a8cf9334c51857397ed50398c8e","value":"Searching best thr : 100%"}},"09a10a8cf9334c51857397ed50398c8e":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"1f3989a0c01248328e16875075e9d1c4":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_08983a9c6aff42578980f4f7113c3ee2","IPY_MODEL_22cfcc0a7cc6455fbf3bb7c788c8a4e1","IPY_MODEL_c8392e8075224e3b8a020a16c1a08447"],"layout":"IPY_MODEL_6cec9a2c2fac450d87248aed8dd62f86"}},"22cfcc0a7cc6455fbf3bb7c788c8a4e1":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_dffe80502d954bdea0bbb6353dbf5515","max":20,"min":0,"orientation":"horizontal","style":"IPY_MODEL_7ce1b34a4f864a42a6619eec82311eb0","value":20}},"4411aefc021d46d0ada7b645eb53ec48":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"6cec9a2c2fac450d87248aed8dd62f86":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7ce1b34a4f864a42a6619eec82311eb0":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"83fe40a0b8f047cc8602206909d42361":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9384babdb7054d55aecdf3e989ddc926":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"c8392e8075224e3b8a020a16c1a08447":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_83fe40a0b8f047cc8602206909d42361","placeholder":"​","style":"IPY_MODEL_9384babdb7054d55aecdf3e989ddc926","value":" 20/20 [04:34&lt;00:00, 12.66s/it]"}},"dffe80502d954bdea0bbb6353dbf5515":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><img src=\"https://keras.io/img/logo-small.png\" alt=\"Keras logo\" width=\"100\"><br/>\nThis starter notebook is provided by the Keras team.</center>","metadata":{"execution":{"iopub.execute_input":"2024-01-10T05:24:31.308329Z","iopub.status.busy":"2024-01-10T05:24:31.307595Z","iopub.status.idle":"2024-01-10T05:24:31.313088Z","shell.execute_reply":"2024-01-10T05:24:31.312113Z","shell.execute_reply.started":"2024-01-10T05:24:31.308287Z"},"papermill":{"duration":0.011755,"end_time":"2024-01-14T03:16:16.447481","exception":false,"start_time":"2024-01-14T03:16:16.435726","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# HMS - Harmful Brain Activity Classification with [KerasCV](https://github.com/keras-team/keras-cv) and [Keras](https://github.com/keras-team/keras)\n\n> The objective of this competition is to classify seizures and other patterns of harmful brain activity in critically ill patients\n\nThis notebook guides you through the process of training and inferring a Deep Learning model, specifically EfficientNetV2, using KerasCV on the competition dataset. Specificaclly, this notebook uses spectrogram of the eeg data to classify the patterns.\n\nFun fact: This notebook is backend-agnostic, supporting TensorFlow, PyTorch, and JAX. Utilizing KerasCV and Keras allows us to choose our preferred backend. Explore more details on [Keras](https://keras.io/keras_core/announcement/).\n\nIn this notebook, you will learn:\n\n* Loading the data efficiently using [`tf.data`](https://www.tensorflow.org/guide/data).\n* Creating the model using KerasCV presets.\n* Training the model.\n* Inference and Submission on test data.\n\n**Note**: For a more in-depth understanding of KerasCV, refer to the [KerasCV guides](https://keras.io/guides/keras_cv/).","metadata":{}},{"cell_type":"markdown","source":"# 🛠 | Install Libraries  \n\nSince internet access is **disabled** during inference, we cannot install libraries in the usual `!pip install <lib_name>` manner. Instead, we need to install libraries from local files. In the following cell, we will install libraries from our local files. The installation code stays very similar - we just use the `filepath` instead of the `filename` of the library. So now the code is `!pip install <local_filepath>`. \n\n> The `filepath` of these local libraries look quite complicated, but don't be intimidated! Also `--no-deps` argument ensures that we are not installing any additional libraries.","metadata":{"papermill":{"duration":0.011416,"end_time":"2024-01-14T03:16:16.470167","exception":false,"start_time":"2024-01-14T03:16:16.458751","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#このコードは、指定されたパスにある特定のホイールファイル（.whl）を使用して、Pythonパッケージをインストールしています。\n#!pip install は、Pythonパッケージをインストールするために使用されるコマンドです\n#-q オプションは、インストールの詳細を非表示にするためのものです。これにより、出力が簡潔になります\n#/kaggle/input/kerasv3-lib-ds/keras_cv-0.8.2-py3-none-any.whl は、インストールするホイールファイルのパスです\n#--no-deps オプションは、依存関係をインストールしないようにするためのものです\n#つまり、このコードは、指定されたホイールファイルを静かに（出力を非表示にして）インストールします\n\n!pip install -q /kaggle/input/kerasv3-lib-ds/keras_cv-0.8.2-py3-none-any.whl --no-deps\n!pip install -q /kaggle/input/kerasv3-lib-ds/tensorflow-2.15.0.post1-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl --no-deps\n!pip install -q /kaggle/input/kerasv3-lib-ds/keras-3.0.4-py3-none-any.whl --no-deps\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install -q keras_cv --upgrade\n#!pip install -q keras_cv==0.8.2\n\n#WARNING: Retrying (Retry(total=4, connect=None, read=None, redirect=None, status=None)) \n#after connection broken by 'NewConnectionError('<pip._vendor.urllib3.connection.HTTPSConnection object at 0x7b751bd114b0>: \n# Failed to establish a new connection: [Errno -3] Temporary failure in name resolution')': /simple/keras-core/\n    \n#この警告は、pip がパッケージをダウンロードしようとしている際に、一時的な名前解決の失敗が発生したことを示しています。\n#名前解決エラーは、通常はネットワーク接続の問題に起因します。","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📚 | Import Libraries ","metadata":{"papermill":{"duration":0.010878,"end_time":"2024-01-14T03:17:49.510159","exception":false,"start_time":"2024-01-14T03:17:49.499281","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#os.environ[\"KERAS_BACKEND\"] = \"jax\" は、KerasのバックエンドをJAXに設定するものです。\n#機械学習では、バックエンドはニューラルネットワークの学習や推論に必要な数学的演算を行う計算エンジンです。\n#Kerasは高レベルのニューラルネットワークAPIであり、ユーザーがTensorFlow、Theano、\n#またはこの場合のJAXなど、異なるバックエンドエンジンを切り替えて使用できます。\n#JAXはGoogleが開発した高性能数値計算ライブラリで、主に機械学習の研究向けに開発されています。\n#NumPyに似たインターフェースを提供しますが、GPU/TPUの高速化や自動微分向けの追加機能を備えており、深層学習のタスクに適しています。\n#os.environ[\"KERAS_BACKEND\"] = \"jax\" を設定することで、\n#Kerasに対してニューラルネットワークの構築や学習にJAXを計算バックエンドとして使用するよう指示します。\n\nimport os\nos.environ[\"KERAS_BACKEND\"] = \"jax\" # you can also use tensorflow or torch\n\nimport keras_cv\nimport keras\nfrom keras import ops\nimport tensorflow as tf\n\nimport cv2\nimport pandas as pd\nimport numpy as np\nfrom glob import glob\nfrom tqdm.notebook import tqdm\nimport joblib\n\nimport matplotlib.pyplot as plt \n\nprint('panda')\n\n\n","metadata":{"papermill":{"duration":10.671979,"end_time":"2024-01-14T03:18:00.193134","exception":false,"start_time":"2024-01-14T03:17:49.521155","status":"completed"},"tags":[],"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#2024-04-12 18:47:28.607960: E external/local_xla/xla/stream_executor/cuda/cuda_dnn.cc:9261] \n#Unable to register cuDNN factory: Attempting to register factory for plugin cuDNN when one has already been registered\n\n#このエラーメッセージは、CUDA関連のライブラリ（cuDNN）の登録に関する問題を示しています。\n#エラーメッセージの内容から、同じプラグインがすでに登録されているにもかかわらず、別のプラグインを登録しようとしていることがわかります。\n#これは、CUDA関連のプラグインが競合していることを示しています。\n\n#この問題を解決するためのいくつかのアプローチがありますが、具体的な解決策は状況によって異なります。一般的なアプローチは以下の通りです：\n\n#ドライバーやCUDA関連のライブラリを更新する： CUDA関連のライブラリやドライバーを最新のバージョンにアップデートします。\n#これにより、潜在的な競合が解消される可能性があります。\n\n#TensorFlowや他のライブラリのバージョンを調整する： \n#使用しているライブラリのバージョンが古い場合、最新のバージョンにアップデートすることで問題が解決される可能性があります。\n#特に、TensorFlowやKerasなどの深層学習ライブラリが関連する問題を引き起こしている可能性があります。\n\n#環境変数を設定する： CUDA関連の環境変数を調整して、競合を解消することもできます。\n#たとえば、TF_FORCE_GPU_ALLOW_GROWTH を設定して、GPUメモリの使用を柔軟に制御することができます。\n\n#これらの方法のいずれかが問題を解決するかどうかは、具体的な状況に依存します。","metadata":{}},{"cell_type":"markdown","source":"## Library Versions","metadata":{"papermill":{"duration":0.010958,"end_time":"2024-01-14T03:18:00.215704","exception":false,"start_time":"2024-01-14T03:18:00.204746","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"TensorFlow:\", tf.__version__)\nprint(\"Keras:\", keras.__version__)\nprint(\"KerasCV:\", keras_cv.__version__)","metadata":{"papermill":{"duration":0.019435,"end_time":"2024-01-14T03:18:00.246368","exception":false,"start_time":"2024-01-14T03:18:00.226933","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ⚙️ | Configuration","metadata":{"papermill":{"duration":0.010922,"end_time":"2024-01-14T03:18:00.26855","exception":false,"start_time":"2024-01-14T03:18:00.257628","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"\"efficientnetv2_b2_imagenet\" は、EfficientNetV2と呼ばれる畳み込みニューラルネットワークアーキテクチャの一部を指します。\n\nEfficientNetは、Googleが提案したモデルアーキテクチャで、高い性能と効率性を両立させることを目指しています。EfficientNetV2はその後のバージョンで、EfficientNetの改良版です。\n\n\"efficientnetv2_b2_imagenet\" は、EfficientNetV2の中で特定のサブモデルを指定するための名前です。この名前は、以下のように解釈できます：\n\n\"efficientnetv2\": EfficientNetV2アーキテクチャであることを示します。\n\"_b2\": EfficientNetV2の中で、中間サイズのモデルであることを示します。このサフィックスは、モデルのスケールを示しており、\"b2\" はその中のサイズを指します。\n\"_imagenet\": このモデルは、ImageNetデータセットで事前学習されていることを示します。ImageNetは、広く使用される大規模な画像データセットであり、多くの画像分類タスクのベンチマークとして使用されます。\nしたがって、\"efficientnetv2_b2_imagenet\" は、EfficientNetV2アーキテクチャの中間サイズのモデルであり、ImageNetデータセットで事前学習されていることを示します。","metadata":{}},{"cell_type":"code","source":"class CFG:\n    verbose = 1  # Verbosity　\n    #ログの冗長性を制御するための変数。デフォルト値は1で、詳細なログを表示します。\n    seed = 42  # Random seed　\n    #ランダムシードを設定するための変数。再現性を確保するために使用されます。\n    preset = \"efficientnetv2_b2_imagenet\"  # Name of pretrained classifier　\n    #事前学習済み分類器の名前を指定するための変数。デフォルト値は \"efficientnetv2_b2_imagenet\" です。\n    image_size = [400, 300]  # Input image size　\n    #入力画像のサイズを指定するための変数。デフォルト値は [400, 300] です。\n    epochs = 13 # Training epochs　\n    #トレーニングエポック数を指定するための変数。デフォルト値は13です。\n    batch_size = 64  # Batch size　\n    #バッチサイズを指定するための変数。デフォルト値は64です。\n    lr_mode = \"cos\" # LR scheduler mode from one of \"cos\", \"step\", \"exp\"　 \n    #学習率スケジューラのモードを指定するための変数。デフォルト値は \"cos\" です。\n    drop_remainder = True  # Drop incomplete batches\n    #不完全なバッチを削除するかどうかを指定するための変数。デフォルト値はTrueです。\n    num_classes = 6 # Number of classes in the dataset\n    #データセット内のクラス数を指定するための変数。デフォルト値は6です。\n    fold = 0 # Which fold to set as validation data\n    #バリデーションデータとして設定するフォールドを指定するための変数。デフォルト値は0です。\n    class_names = ['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other']\n    #クラスの名前のリストを指定するための変数。\n    label2name = dict(enumerate(class_names))\n    #{0: 'Seizure', 1: 'LPD', 2: 'GPD', 3: 'LRDA', 4: 'GRDA', 5: 'Other'}\n    #クラスのラベルからクラス名へのマッピングを示す辞書\n    name2label = {v:k for k, v in label2name.items()}\n    #クラス名からクラスのラベルへのマッピングを示す辞書。\n    #{'Seizure': 0, 'LPD': 1, 'GPD': 2, 'LRDA': 3, 'GRDA': 4, 'Other': 5}\n\nprint(\"panda\")","metadata":{"papermill":{"duration":0.018795,"end_time":"2024-01-14T03:18:00.298534","exception":false,"start_time":"2024-01-14T03:18:00.279739","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ♻️ | Reproducibility \nSets value for random seed to produce similar result in each run.","metadata":{"papermill":{"duration":0.010907,"end_time":"2024-01-14T03:18:00.32063","exception":false,"start_time":"2024-01-14T03:18:00.309723","status":"completed"},"tags":[]}},{"cell_type":"code","source":"keras.utils.set_random_seed(CFG.seed)\nprint('panda')","metadata":{"papermill":{"duration":0.018371,"end_time":"2024-01-14T03:18:00.350074","exception":false,"start_time":"2024-01-14T03:18:00.331703","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📁 | Dataset Path ","metadata":{"papermill":{"duration":0.010888,"end_time":"2024-01-14T03:18:00.372053","exception":false,"start_time":"2024-01-14T03:18:00.361165","status":"completed"},"tags":[]}},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\n\nSPEC_DIR = \"/tmp/dataset/hms-hbac\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\nos.makedirs(SPEC_DIR+'/test_spectrograms', exist_ok=True)\n\nprint('panda')\n#具体的には、SPEC_DIR+'/train_spectrograms' で指定されたパスに対してディレクトリを作成します。\n#exist_ok=True は、既にそのディレクトリが存在する場合にエラーを発生させずに処理を継続することを指定します。\n#つまり、指定されたディレクトリが存在しない場合にのみ作成を試み、既に存在する場合は何もしません。","metadata":{"papermill":{"duration":0.017704,"end_time":"2024-01-14T03:18:00.400852","exception":false,"start_time":"2024-01-14T03:18:00.383148","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📖 | Meta Data ","metadata":{"papermill":{"duration":0.011434,"end_time":"2024-01-14T03:18:00.472401","exception":false,"start_time":"2024-01-14T03:18:00.460967","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Train + Valid\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf['eeg_path'] = f'{BASE_PATH}/train_eegs/'+df['eeg_id'].astype(str)+'.parquet'\n#DataFrame df の中に新しい列 eeg_path を追加する操作です。\n#具体的には、df['eeg_id'] 列の各要素を文字列に変換し、.parquet の拡張子を付けた後、\n#それぞれの要素の前に '{BASE_PATH}/train_eegs/' を追加します。そして、その結果を新しい列 eeg_path として DataFrame に追加します。\n\n\ndf['spec_path'] = f'{BASE_PATH}/train_spectrograms/'+df['spectrogram_id'].astype(str)+'.parquet'\ndf['spec2_path'] = f'{SPEC_DIR}/train_spectrograms/'+df['spectrogram_id'].astype(str)+'.npy'\ndf['class_name'] = df.expert_consensus.copy()\n#DataFrame df の expert_consensus 列の値をコピーし、新しい列 class_name として追加する操作です。\ndf['class_label'] = df.expert_consensus.map(CFG.name2label)\n#例えば、df.expert_consensus 列が ['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA', 'Other'] であり、\n#CFG.name2label 辞書が {'Seizure': 0, 'LPD': 1, 'GPD': 2, 'LRDA': 3, 'GRDA': 4, 'Other': 5} の場合、\n#df['class_label'] 列は [0, 1, 2, 3, 4, 5] となります。\ndisplay(df.head(2))\n\n# Test\ntest_df = pd.read_csv(f'{BASE_PATH}/test.csv')\ntest_df['eeg_path'] = f'{BASE_PATH}/test_eegs/'+test_df['eeg_id'].astype(str)+'.parquet'\ntest_df['spec_path'] = f'{BASE_PATH}/test_spectrograms/'+test_df['spectrogram_id'].astype(str)+'.parquet'\ntest_df['spec2_path'] = f'{SPEC_DIR}/test_spectrograms/'+test_df['spectrogram_id'].astype(str)+'.npy'\ndisplay(test_df.head(2))\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert `.parquet` to `.npy`\n\nTo facilitate easier data loading, we will convert the EEG spectrograms from `parquet` to `npy` format. This process involves saving the spectrogram data, and since the content of the files remains the same, no significant changes are made. \n\n> It's worth noting that the `time` column is excluded, as it is not part of the spectrogram.","metadata":{}},{"cell_type":"code","source":"#このコードは、単一の EEG データを処理するための関数 process_spec を定義しています。\n# Define a function to process a single eeg_id\ndef process_spec(spec_id, split=\"train\"):\n    \n    spec_path = f\"{BASE_PATH}/{split}_spectrograms/{spec_id}.parquet\"\n    #引数 spec_id に基づいて、特定の EEG データのパスを生成します。\n    #パスは BASE_PATH と split（デフォルトは \"train\"）に基づいています。\n    # \"..../train__spectrograms/123.parquet\"\n    \n    spec = pd.read_parquet(spec_path)\n    #生成されたパスを使用して、Parquet形式のファイルからデータを読み込みます。\n    \n    spec = spec.fillna(0).values[:, 1:].T # fill NaN values with 0, transpose for (Time, Freq) -> (Freq, Time)\n    #spec.fillna(0) - 読み込んだデータに対して、NaN（欠損値）を0で埋めます。\n    #values[:, 1:].T - そして、データを配列に変換し、転置します。\n    #これにより、データの形状が (Time, Freq) から (Freq, Time) に変更されます。\n    \n    spec = spec.astype(\"float32\")\n    #変換されたデータを浮動小数点数の配列として保存します。\n    \n    np.save(f\"{SPEC_DIR}/{split}_spectrograms/{spec_id}.npy\", spec)\n    #保存先のパスは SPEC_DIR と split に基づいています。\n\n    \n#train__spectrograms/123.npy created and saved\nprint('panda')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get unique spec_ids of train and valid data\nspec_ids = df[\"spectrogram_id\"].unique()\n#DataFrame `df` の中の `spectrogram_id` 列から一意な値を抽出し、それらの値を配列 `spec_ids` に格納する操作です。\n\n#具体的には、`df[\"spectrogram_id\"].unique()` は `spectrogram_id` 列の値の一意なセットを取得します。\n#そして、それらの一意な値を配列として返します。`unique()` 関数は、重複した値を除外し、一意な値のみを抽出します。\n#この操作により、`spec_ids` 配列には `spectrogram_id` 列の中で一意な値が含まれ、それらの値が重複せずに格納されます。\n#これは、特定のデータフレーム列内のユニークな識別子やカテゴリーを取得するために使用されます。\n\n'''\n  spectrogram_id  value\n0               1     10\n1               2     20\n2               1     30\n3               3     40\n4               2     50\n'''\n#value_for_id_1 = df[df[\"spectrogram_id\"] == 1][\"value\"]\n#これにより、spectrogram_id 列が1に対応する行の value 列が取得されます。\n#例えば、上記のDataFrameの場合、value_for_id_1 は [10, 30] という値を持ちます。\n\n\n\n\n# Parallelize the processing using joblib for training data\n_ = joblib.Parallel(n_jobs=-1, backend=\"loky\")(\n    joblib.delayed(process_spec)(spec_id, \"train\")\n    for spec_id in tqdm(spec_ids, total=len(spec_ids))\n)\n\n# Get unique spec_ids of test data\ntest_spec_ids = test_df[\"spectrogram_id\"].unique()\n\n# Parallelize the processing using joblib for test data\n_ = joblib.Parallel(n_jobs=-1, backend=\"loky\")(\n    joblib.delayed(process_spec)(spec_id, \"test\")\n    for spec_id in tqdm(test_spec_ids, total=len(test_spec_ids))\n)\n\n#call process_spec to create and save npy files\nprint('panda')","metadata":{"papermill":{"duration":0.86264,"end_time":"2024-01-14T03:18:01.346487","exception":false,"start_time":"2024-01-14T03:18:00.483847","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🍚 | DataLoader\n\nThis DataLoader first reads `npy` spectrogram files and extracts labeled subsamples using specified `offset` values. Then, it converts the spectrogram data into `log spectrogram` and applies the popular signal augmentation `MixUp`.\n\n> Note that, we are converting the mono channel signal to a 3-channel signal for using \"ImageNet\" weights of pretrained model.","metadata":{"papermill":{"duration":0.011843,"end_time":"2024-01-14T03:18:01.457956","exception":false,"start_time":"2024-01-14T03:18:01.446113","status":"completed"},"tags":[]}},{"cell_type":"code","source":"#データ拡張機能を構築するための関数 build_augmenter を定義しています。\n\n#dim パラメータによって指定された画像のサイズ（デフォルトは CFG.image_size）を元に、\n#データ拡張のためのオーグメンテーションレイヤーを定義します。\n#ここでは、 #MixUpやランダムカットアウト（マスキング）のオーグメンテーションが定義されています。\n\n\ndef build_augmenter(dim=CFG.image_size):\n    augmenters = [\n        keras_cv.layers.MixUp(alpha=2.0),\n       \n        keras_cv.layers.RandomCutout(height_factor=(1.0, 1.0),\n                                     width_factor=(0.06, 0.1)), # freq-masking\n        keras_cv.layers.RandomCutout(height_factor=(0.06, 0.1),\n                                     width_factor=(1.0, 1.0)), # time-masking\n    ]\n    \n#このコードは、データ拡張を行うための関数 `augment` を定義しています。\n#この関数は、入力として画像とラベルのペアを受け取り、それらに対してデータ拡張を適用して変換された画像とラベルのペアを返します。\n    def augment(img, label):\n    #関数 `augment` を定義します。この関数は、画像 (`img`) とラベル (`label`) のペアを引数として受け取ります。\n        \n        data = {\"images\":img, \"labels\":label}\n        #データを格納するための辞書 `data` を作成します。\n        #この辞書には、キー `\"images\"` に画像 (`img`) を、キー `\"labels\"` にラベル (`label`) を格納します。\n        \n        for augmenter in augmenters:\n        #データ拡張を行うための各オーグメンテーションレイヤーについて、ループを行います。\n        #`augmenters` は、事前に定義されたデータ拡張のリストです。\n         \n            if tf.random.uniform([]) < 0.5:\n            #データ拡張を適用するかどうかを決定するための条件です。\n            #ここでは、`tf.random.uniform([])` を使用して0から1までの一様分布からランダムな数を生成し、\n            #それが0.5未満であるかどうかを確認しています。\n            #つまり、50%の確率でデータ拡張を適用します。\n            \n            #`tf.random.uniform([])` は、TensorFlowのランダムな一様分布からサンプリングされた値を返します。\n            #引数 `[]` は、生成されるテンソルの形状を示しており、空のリスト `[]` を指定することでスカラー値を生成します。\n            #つまり、`tf.random.uniform([])` は、0から1までの範囲で一様分布からランダムに選ばれた値を返します。\n                \n                \n                data = augmenter(data, training=True)\n                #データ拡張を適用します。`augmenter` は、ループ内で定義された各データ拡張のオブジェクトです。\n                #`data` 辞書と `training=True` を渡してデータ拡張を適用し、変換されたデータを `data` に上書きします。\n                \n        return data[\"images\"], data[\"labels\"]\n        #変換された画像とラベルのペアを返します。\n        #`data` 辞書の `\"images\"` キーと `\"labels\"` キーの値をそれぞれ返します。\n    \n    return augment\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n\ndef build_decoder(with_labels=True, target_size=CFG.image_size, dtype=32):\n    def decode_signal(path, offset=None):\n        # Read .npy files and process the signal\n        file_bytes = tf.io.read_file(path)\n        sig = tf.io.decode_raw(file_bytes, tf.float32)\n        sig = sig[1024//dtype:]  # Remove header tag\n        sig = tf.reshape(sig, [400, -1])\n        \n        # Extract labeled subsample from full spectrogram using \"offset\"\n        if offset is not None: \n            offset = offset // 2  # Only odd values are given\n            sig = sig[:, offset:offset+300]\n            \n            # Pad spectrogram to ensure the same input shape of [400, 300]\n            pad_size = tf.math.maximum(0, 300 - tf.shape(sig)[1])\n            sig = tf.pad(sig, [[0, 0], [0, pad_size]])\n            sig = tf.reshape(sig, [400, 300])\n        \n        # Log spectrogram \n        sig = tf.clip_by_value(sig, tf.math.exp(-4.0), tf.math.exp(8.0)) # avoid 0 in log\n        sig = tf.math.log(sig)\n        \n        # Normalize spectrogram\n        sig -= tf.math.reduce_mean(sig)\n        sig /= tf.math.reduce_std(sig) + 1e-6\n        \n        # Mono channel to 3 channels to use \"ImageNet\" weights\n        sig = tf.tile(sig[..., None], [1, 1, 3])\n        return sig\n    \n    def decode_label(label):\n        label = tf.one_hot(label, CFG.num_classes)\n        label = tf.cast(label, tf.float32)\n        label = tf.reshape(label, [CFG.num_classes])\n        return label\n    \n    def decode_with_labels(path, offset=None, label=None):\n        sig = decode_signal(path, offset)\n        label = decode_label(label)\n        return (sig, label)\n    \n    return decode_with_labels if with_labels else decode_signal\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef build_dataset(paths, \n                  offsets=None, \n                  labels=None, \n                  batch_size=32, \n                  cache=True,\n                  decode_fn=None, \n                  augment_fn=None,\n                  augment=False, \n                  repeat=True, \n                  shuffle=1024, \n                  cache_dir=\"\", \n                  drop_remainder=False):\n    \n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter()\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = (paths, offsets) if labels is None else (paths, offsets, labels)\n    \n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=AUTO)\n    ds = ds.cache(cache_dir) if cache else ds\n    ds = ds.repeat() if repeat else ds\n    if shuffle: \n        ds = ds.shuffle(shuffle, seed=CFG.seed)\n        opt = tf.data.Options()\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n    ds = ds.batch(batch_size, drop_remainder=drop_remainder)\n    ds = ds.map(augment_fn, num_parallel_calls=AUTO) if augment else ds\n    ds = ds.prefetch(AUTO)\n    return ds\n\nprint('panda')","metadata":{"papermill":{"duration":0.039133,"end_time":"2024-01-14T03:18:01.509017","exception":false,"start_time":"2024-01-14T03:18:01.469884","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🔪 | Data Split\n\nIn the following code snippet, the data is divided into `5` folds. Note that, the `groups` argument is used to prevent any overlap of patients between the training and validation sets, thus avoiding potential **data leakage** issues. Additionally, each split is stratified based on the `class_label`, ensuring a uniform distribution of class labels in each fold.","metadata":{"papermill":{"duration":0.012174,"end_time":"2024-01-14T03:18:01.538524","exception":false,"start_time":"2024-01-14T03:18:01.52635","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedGroupKFold\n\nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=CFG.seed)\n\ndf[\"fold\"] = -1\ndf.reset_index(drop=True, inplace=True)\nfor fold, (train_idx, valid_idx) in enumerate(\n    sgkf.split(df, y=df[\"class_label\"], groups=df[\"patient_id\"])\n):\n    df.loc[valid_idx, \"fold\"] = fold\ndf.groupby([\"fold\", \"class_name\"])[[\"eeg_id\"]].count().T\n\nprint('panda')","metadata":{"papermill":{"duration":0.037496,"end_time":"2024-01-14T03:18:01.587924","exception":false,"start_time":"2024-01-14T03:18:01.550428","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build Train & Valid Dataset\n\nOnly first sample for each `spectrogram_id` is used in order to keep the dataset size managable. Feel free to train on full data.","metadata":{"papermill":{"duration":0.011875,"end_time":"2024-01-14T03:18:01.611955","exception":false,"start_time":"2024-01-14T03:18:01.60008","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Sample from full data\nsample_df = df.groupby(\"spectrogram_id\").head(1).reset_index(drop=True)\n#DataFrame `df` を`spectrogram_id` 列でグループ化し、\n#それぞれのグループから最初の行だけを取得して新しいDataFrame `sample_df` を作成します。\n#そして、`sample_df` のインデックスをリセットし、元のインデックスを削除します。\n\n#具体的には、以下の手順が行われます：\n#1. `df.groupby(\"spectrogram_id\")`：`df` を `spectrogram_id` 列でグループ化します。\n#つまり、同じ `spectrogram_id` を持つ行が1つのグループになります。\n\n#2. `.head(1)`：各グループから最初の行だけを取得します。つまり、各 `spectrogram_id` ごとに最初の行だけが残ります。\n\n#3. `.reset_index(drop=True)`：インデックスをリセットします。`drop=True` は、元のインデックスを削除することを意味します。\n#リセットされたインデックスは、0から始まる新しい連番になります。\n\n#これにより、DataFrame `sample_df` には、元のDataFrame `df` の各 `spectrogram_id` ごとに1つの行だけが残り、\n#インデックスがリセットされた新しいDataFrame が作成されます。\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#はい、DataFrame の中で特定の列の値が重複しているかどうかを調べることができます。\n\n#具体的には、`duplicated()` メソッドを使用して特定の列の値が重複している行をブール値で取得し、それをさらに `any()` メソッドで調べます。\n\n#例えば、`spectrogram_id` 列が重複している行が DataFrame `df` にあるかどうかを調べるには、次のようにします：\n\n\n#duplicated_rows = sample_df.duplicated(subset=['spectrogram_id'], keep=False)\n#ここで、`subset=['spectrogram_id']` は、`spectrogram_id` 列を対象に重複を調べることを指定します。\n#`keep=False` は、重複しているすべての行を True として返すことを指定します。\n\n#print(df.shape) #(106800, 21)\n#print(sample_df.shape) #(11138, 21)\n#print(duplicated_rows.shape) #(11138,)\n\n#any_duplicated = duplicated_rows.any()\n#`any_duplicated` には、`spectrogram_id` 列が重複している行が少なくとも 1 つ以上ある場合に True が返されます。\n\n#print(any_duplicated.shape) #()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n\n\ntrain_df = sample_df[sample_df.fold != CFG.fold] # CFG.fold = 0\n#DataFrame `sample_df` の中から `fold` 列の値が `CFG.fold` と異なる行だけを選択して、\n#新しいDataFrame `train_df` を作成します。\n\n#具体的には、次の手順が行われます：\n\n#1. `sample_df.fold != CFG.fold`：  # CFG.fold = 0\n#`sample_df` の `fold` 列の値が `CFG.fold` と異なる行を示すブール値のシリーズが生成されます。\n#これは、各行が `CFG.fold` と異なる場合に `True` を、同じ場合に `False` を持ちます。\n\n#2. `sample_df[sample_df.fold != CFG.fold]`：\n#上記で生成したブール値のシリーズを使用して、DataFrame `sample_df` から条件に合致する行だけを選択します。\n#つまり、`fold` 列の値が `CFG.fold` と異なる行だけが残ります。\n\n#3. `train_df = ...`：選択された行が新しいDataFrame `train_df` として保存されます。\n\n#したがって、`train_df` には `fold` 列の値が `CFG.fold` と異なる行のみが含まれる新しいDataFrame が作成されます。\n#通常、このような操作は、交差検証のためのトレーニングデータを準備する際に使用されます。\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df['fold'].unique()) #array([2, 3, 1, 4]) \n#0 以外\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(train_df)) #<class 'pandas.core.frame.DataFrame'>\nprint(train_df.shape) #(9166, 21) #fold=0 は入ってない\n#train_df\n\n'''\nprint(train_df.columns)\nIndex(['eeg_id', 'eeg_sub_id', 'eeg_label_offset_seconds', \n       'spectrogram_id','spectrogram_sub_id', 'spectrogram_label_offset_seconds', \n       'label_id','patient_id', \n       'expert_consensus', \n       'seizure_vote', 'lpd_vote','gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote', \n       'eeg_path','spec_path', 'spec2_path', \n       'class_name', 'class_label', 'fold'],\n      dtype='object')\n'''\n#---------------------\n#eeg_id\n\n#eeg_sub_id\t\n#print(train_df['eeg_sub_id'].unique()) #[0]\n\n#eeg_label_offset_seconds\n#print(train_df['eeg_label_offset_seconds'].unique()) #[0.]\n\n#spectrogram_id\t\n\n#spectrogram_sub_id\t\n#print(train_df['spectrogram_sub_id'].unique()) #[0]\n\n#spectrogram_label_offset_seconds\n#print(train_df['spectrogram_label_offset_seconds'].unique()) #[0.]\n\n#label_id\n#print(train_df['label_id'].unique()) #[1978807404  557980729 1963161945 ... 2394260563 1216355904  429140316]\n\n#patient_id\t\n\n#expert_consensus\n#print(train_df['expert_consensus'].unique()) #['GPD' 'LRDA' 'Seizure' 'Other' 'GRDA' 'LPD']\n\n#seizure_vote\t...\tgpd_vote\tlrda_vote\tgrda_vote\tother_vote\t\n#eeg_path   - parquet path\n#spec_path  - parquet path\n#spec2_path  - nyp path\n\n#class_name\n#print(train_df['class_name'].unique()) #['GPD' 'LRDA' 'Seizure' 'Other' 'GRDA' 'LPD']\n\n#class_label  \n#print(train_df['class_label'].unique()) #[2 3 0 5 4 1]\n\n#fold  \n#print(train_df['fold'].unique()) #[2 3 1 4]\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train\ntrain_paths = train_df.spec2_path.values\n#npy file path\n\nprint(type(train_paths))#<class 'numpy.ndarray'>\nprint(train_paths.shape) #(9166,)\n\ntrain_offsets = train_df.spectrogram_label_offset_seconds.values.astype(int)\n\nprint(type(train_offsets))#<class 'numpy.ndarray'>\nprint(train_offsets.shape)#(9166,)\n\ntrain_labels = train_df.class_label.values\n\nprint(type(train_labels)) #<class 'numpy.ndarray'>\nprint(train_labels.shape) #(9166,)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_paths[0]) #/tmp/dataset/hms-hbac/train_spectrograms/924234.npy\nprint(train_offsets[0]) #0\nprint(train_labels[0]) #2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths=train_paths\noffsets=train_offsets \nlabels=train_labels \nbatch_size=CFG.batch_size \ncache=True\ndecode_fn=None \naugment_fn=None\naugment=True \nrepeat=True \nshuffle=True \ncache_dir=\"\"\ndrop_remainder=False\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if cache_dir != \"\" and cache is True:\n    print('make...') #not make\n    os.makedirs(cache_dir, exist_ok=True)\n\n'''\nこのコードは、`cache_dir` が空文字列でなく、かつ `cache` フラグがTrueである場合に、指定されたディレクトリを作成します。\n\n具体的には、以下の手順が行われます：\n\n1. `cache_dir != \"\" and cache is True`：\n`cache_dir` が空文字列でなく、かつ `cache` フラグがTrueであるかどうかを確認します。\nこれは、条件が両方とも満たされた場合にTrueとなります。\n\n2. `os.makedirs(cache_dir, exist_ok=True)`：\n条件がTrueの場合、`os.makedirs()` 関数を使用して `cache_dir` で指定されたディレクトリを作成します。\n`exist_ok=True` は、ディレクトリが既に存在している場合にエラーを発生させずに処理を続行するためのオプションです。\n\nしたがって、このコードは指定されたディレクトリが存在しない場合にそのディレクトリを作成する役割を果たします。    \n'''     \n\n#does not make cache dir\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if decode_fn is None:\n    print('yes') #yes\n    decode_fn = build_decoder(labels is not None)\n\n'''\nこのコードは、`decode_fn` が指定されていない場合に、デコード関数 `decode_fn` を構築します。\n\n具体的には、以下の手順が行われます：\n\n1. `if decode_fn is None:`：\n`decode_fn` が指定されていないかどうかをチェックします。\n`is None` は、`decode_fn` が None（未定義）であるかどうかを確認します。\n\n2. `decode_fn = build_decoder(labels is not None)`：\n`build_decoder` 関数を呼び出してデコード関数を構築します。\n`labels is not None` の部分は、`labels` が指定されているかどうかを確認します。\nもし `labels` が None でない場合、つまりラベルが与えられている場合は `True` を返します。\nそれ以外の場合は `False` を返します。\nこの値が `build_decoder` 関数に渡され、適切なデコード関数が構築されます。\n\nつまり、このコードは、デコード関数が指定されていない場合に、ラベルが与えられているかどうかに応じて適切なデコード関数を構築します。\n'''\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if augment_fn is None:\n    print('yes')#YES\n    augment_fn = build_augmenter()\n\n    \nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n\n'''\n`tf.data.experimental.AUTOTUNE` は、\nTensorFlowのデータセットパフォーマンスチューニング機能を利用するための特別な値です。\n\nTensorFlowのデータパイプラインでは、データの読み込み、前処理、およびバッチ処理などの操作があります。\nこれらの操作は、CPUやGPUなどのリソースを効率的に利用するために最適化することができます。\n`AUTOTUNE` を使用すると、TensorFlowが自動的に最適なパフォーマンスを得るための最適な実行設定を選択します。\n\n具体的には、`AUTOTUNE` をデータセットのパフォーマンス設定に指定することで、\nTensorFlowはバックグラウンドでデータの読み込み、前処理、およびバッチ処理の並行実行を最適化します。\nこれにより、データパイプラインの処理速度が向上し、モデルのトレーニングや推論の効率が向上します。\n\n例えば、次のように使用されます：\n\n```python\ndataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n```\n\nこのようにすることで、データセットの次のバッチの読み込みがモデルのトレーニングと同時に行われ、\nデータの読み込みの待ち時間を削減することができます。\n'''\n\nprint(type(AUTO)) #\nprint(AUTO) #-1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"slices = (paths, offsets) if labels is None else (paths, offsets, labels)\n\n'''\nこのコードは、Pythonの条件式を使用して、タプル `slices` を作成しています。\n\n具体的には、以下の手順が行われます：\n\n1. `labels is None`：`labels` が None（未定義）かどうかをチェックします。\n\n2. `slices = (paths, offsets) if labels is None else (paths, offsets, labels)`：\nPythonの条件式を使用して、タプル `slices` を作成します。\nもし `labels` が None であれば、`(paths, offsets)` を要素とするタプルを作成します。\nそうでない場合は、`(paths, offsets, labels)` を要素とするタプルを作成します。\n\nつまり、このコードは、`labels` が指定されている場合とされていない場合で異なる構造のタプルを作成します。\n条件によって、要素の数が変化することに注意してください。\n'''\n\nprint(type(slices)) #<class 'tuple'>\nprint(len(slices)) #3\nprint(type(slices[0])) #<class 'numpy.ndarray'>\nprint(slices[0].shape) #(9166,)\nprint(slices[0][0]) # npy file path = /tmp/dataset/hms-hbac/train_spectrograms/924234.npy\nprint(slices[1][0]) # offset = 0\nprint(slices[2][0]) # label = 2\n\n\n'''\nslices\n\ndata 1        data 2     data 3 ....\nnpyfile1    npyfile2\noffset1     offset2\nlabel1      label2\n'''\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = tf.data.Dataset.from_tensor_slices(slices)\n\n'''\nこのコードは、与えられたタプル `slices` を入力として受け取り、\nTensorFlowのデータセットオブジェクトを作成します。\n\n具体的には、`from_tensor_slices` メソッドを使用して、\n入力されたタプル `slices` の各要素をテンソルとして扱います。\nこれにより、データセットは各要素のタプルの形状に応じて適切に分割されます。\n\nつまり、`slices` タプルの各要素がそれぞれ `paths`、`offsets`、および `labels` である場合、\n`ds` は以下のようなデータセットになります：\n\n- 各サンプルは `(path, offset)` または `(path, offset, label)` のタプルで構成されます。\n- サンプル数は `slices` の最初の要素（`paths`）の長さに等しくなります。\n\nしたがって、`ds` は与えられた入力データを、TensorFlowのデータセットとして扱える形式に変換します。\n'''\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(ds)) #<class 'tensorflow.python.data.ops.from_tensor_slices_op._TensorSliceDataset'>\n\n\n'''\n`ds` がTensorFlowのデータセットオブジェクトである場合、その中身を調べるには \n`take()` メソッドや `as_numpy_iterator()` メソッドなどを使用して、データセットから要素を取得することができます。\n\n例えば、最初の数個の要素を取得する場合は `take()` メソッドを使用します。次のようにします：\n\n'''\n# 最初の要素だけを取得する例\nfirst_item = next(iter(ds.take(1)))\nprint(first_item)\n\n(<tf.Tensor: \n shape=(), \n dtype=string, \n numpy=b'/tmp/dataset/hms-hbac/train_spectrograms/924234.npy'>, \n \n <tf.Tensor: shape=(), dtype=int64, numpy=0>, <tf.Tensor: shape=(), dtype=int64, numpy=2>)\n\n\n'''\n# 最初の5つの要素を取得する例\nfor item in ds.take(1):\n    print(item)\n'''\n    \n'''\nまたは、データセットのすべての要素をイテレートして取得する場合は `as_numpy_iterator()` メソッドを使用します。次のようにします：\n\n```python\n# データセットのすべての要素を取得する例\nfor item in ds.as_numpy_iterator():\n    print(item)\n```\n\nこれらの方法でデータセットの中身を調べることができます。データセット内の各要素は、タプルやテンソルなど、元の入力データの形式に従います。\n'''\n\n\nprint('panda')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"3//2","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#前のあるやつ\ndef build_decoder(with_labels=True, target_size=CFG.image_size, dtype=32):\n'''\nこの関数 `build_decoder` は、データセット内の各要素をデコードするためのデコード関数を構築します。\nこの関数は、オプションでラベルを含むかどうか、目標のサイズ、およびデータの型を指定することができます。\n\n具体的には、以下のパラメータが定義されています：\n\n- `with_labels=True`：ラベルを含むかどうかを指定します。デフォルトではラベルを含むように設定されています。\n- `target_size=CFG.image_size`：デコードされる信号の目標のサイズを指定します。デフォルトでは `CFG.image_size` の値が使用されます。\n- `dtype=32`：デコードされるデータの型を指定します。デフォルトでは `32`（float32）の値が使用されます。\n\nこの関数は、内部で3つのサブ関数を定義しています：\n\n1. `decode_signal`：信号をデコードするための関数です。この関数は、パスから信号を読み込み、適切な形式に変換します。\n2. `decode_label`：ラベルをデコードするための関数です。この関数は、ラベルをone-hotベクトルに変換します。\n3. `decode_with_labels`：信号とラベルの両方をデコードするための関数です。\nこの関数は、`decode_signal` と `decode_label` を使用して信号とラベルをデコードします。\n\n最後に、`build_decoder` 関数は、`with_labels` パラメータに基づいて、\nラベルを含むかどうかに応じて `decode_with_labels` または `decode_signal` を返します。\n\nこれにより、データセット内の各要素をデコードするための適切なデコード関数が構築されます。\n'''\n   \n    \n    def decode_signal(path, offset=None):\n        \n        # Read .npy files and process the signal\n        file_bytes = tf.io.read_file(path)\n        # read in .npy file\n        #file_bytes is TensorFlow's  tensor object\n        '''\n        `tf.io.read_file(path)` は、指定されたパスにあるファイルをバイト列として読み込む TensorFlow の関数です。\n\n具体的には、次のことが行われます：\n\n- `path` パラメータは、読み込むファイルのパスを指定します。\n- 関数が呼び出されると、指定されたパスのファイルが開かれ、その内容がバイト列として読み込まれます。\n- 読み込まれたバイト列は、TensorFlow のテンソルオブジェクトとして返されます。\nこのテンソルは、データ型が `tf.string` であり、バイト列がそのまま要素として含まれます。\n\nこの関数を使うことで、ファイルの内容をバイト列として読み込んで TensorFlow で処理することができます。\n        '''\n        \n        '''\n        バイト列とは、コンピュータがデータを扱う際に使用する基本的なデータ表現形式の一つです。\n        バイト列は、0から255までの整数値の配列で、1バイトごとに1つの整数値が格納されます。\n        これにより、さまざまな種類のデータ（テキスト、画像、音声など）をバイナリ形式で表現することができます。\n\nバイト列としてファイルを読み込むことは、ファイルの内容をそのままバイナリデータとしてメモリに読み込むことを意味します。\nこれにより、ファイル内の情報を直接操作したり、別のファイルに書き出したりすることが可能になります。\n特に、画像や音声などのバイナリデータを扱う場合には、バイト列として読み込むことが一般的です。\n        '''\n        \n        '''\n        はい、その通りです。バイト（byte）は、8ビットの固定長のデータ単位であり、通常は0から255までの整数値を表します。\n        バイト列は、これらのバイトが連続して配置されたデータのシーケンスを指します。\n        コンピュータ内では、テキスト、画像、音声などのあらゆる種類のデータがバイト列として表現されます。\n        ファイルをバイト列として読み込むことは、ファイル内のデータをメモリにロードし、\n        バイナリデータとしてプログラム内で処理できるようにすることを意味します。\n        '''\n        \n        sig = tf.io.decode_raw(file_bytes, tf.float32)\n        '''\n        `tf.io.decode_raw()` 関数は、バイト列を指定されたデータ型のテンソルにデコードするための TensorFlow の関数です。\n\n具体的には、次のことが行われます：\n\n- `file_bytes` パラメータは、デコードするバイト列を表します。これは、`tf.string` 型のテンソルとして提供されます。\n- `tf.float32` パラメータは、デコードされたテンソルのデータ型を指定します。\nここでは、データ型を float32 として指定しています。これは、デコードされたテンソルの各要素が float32 型の値になることを意味します。\n\n関数が呼び出されると、バイト列が指定されたデータ型のテンソルにデコードされます。\nバイト列内のデータは、指定されたデータ型に応じて適切に解釈され、対応するテンソルに格納されます。\nこの場合、バイト列内のデータは float32 の値として解釈され、float32 のテンソルに格納されます。\n        '''\n        \n        \n        dtype=32\n        sig = sig[1024//dtype:]  # Remove header tag\n        '''\n問題ありません。`dtype=32` ではなく、`dtype` は通常データ型を示すものであり、ここでは float32 を指定しています。\n\nしたがって、`1024//dtype` は `1024//32` となり、32 ビットのデータ型の場合、データの先頭から 32 バイトを除去することを意味します。\n\nつまり、`sig = sig[1024//dtype:]` は、デコードされた信号データから先頭の 32 バイトを除去し、その後のデータを保持するための操作を行っています。\nこれにより、不要なヘッダータグなどが除去され、信号データが正確な位置から始まるようになります。\n        '''\n        sig = tf.reshape(sig, [400, -1])\n        \n        '''\n        `tf.reshape(sig, [400, -1])` は、テンソル `sig` の形状を変更する TensorFlow の関数です。\n\n具体的には、次のようなことが行われます：\n\n- `sig`：\n変形する対象のテンソルです。ここでは、デコードされた信号データが格納されています。\n\n- `[400, -1]`：変形後の形状を示すリストです。\n第1次元のサイズが 400 であり、第2次元のサイズは自動的に決定されます（-1 は自動的に決定されることを示す）。\n\nこの関数を使うと、`sig` テンソルの形状を `(400, -1)` に変更できます。\n第1次元のサイズが 400 になり、第2次元のサイズはデータの長さに応じて自動的に決定されます。\nしたがって、この操作により、信号データが 400 行の行列に変形されます。\n        '''\n        \n        \n        # Extract labeled subsample from full spectrogram using \"offset\"\n        if offset is not None: \n            \n            \n            offset = offset // 2  # Only odd values are given\n            #   3//2 = 1  小数点切り捨て\n         \n            sig = sig[:, offset:offset+300]\n            \n            '''\n            `sig = sig[:, offset:offset+300]` は、2次元テンソル `sig` のスライシング操作を行っています。\n\n具体的には、次のようなことが行われます：\n\n- `sig`：スライスを行う対象の2次元テンソルです。ここでは、デコードされた信号データが格納されています。\n\n- `[:, offset:offset+300]`：行列のスライス範囲を指定しています。`:` は行方向（すべての行）を示し、\n`offset:offset+300` は列方向のスライス範囲を示します。\n具体的には、オフセット値から始まる列から、300 列分のデータを取得します。\n\nしたがって、この操作により、2次元テンソル `sig` からオフセット値から始まる 300 列の部分テンソルが取得されます。\nこれにより、オフセット値から始まる部分のデータが抽出され、300 列にわたって取得されます。\n            '''\n            \n            \n            # Pad spectrogram to ensure the same input shape of [400, 300]\n            pad_size = tf.math.maximum(0, 300 - tf.shape(sig)[1])\n            sig = tf.pad(sig, [[0, 0], [0, pad_size]])\n            sig = tf.reshape(sig, [400, 300])\n        \n        # Log spectrogram \n        sig = tf.clip_by_value(sig, tf.math.exp(-4.0), tf.math.exp(8.0)) # avoid 0 in log\n        sig = tf.math.log(sig)\n        \n        # Normalize spectrogram\n        sig -= tf.math.reduce_mean(sig)\n        sig /= tf.math.reduce_std(sig) + 1e-6\n        \n        # Mono channel to 3 channels to use \"ImageNet\" weights\n        sig = tf.tile(sig[..., None], [1, 1, 3])\n        return sig\n    \n    def decode_label(label):\n        label = tf.one_hot(label, CFG.num_classes)\n        label = tf.cast(label, tf.float32)\n        label = tf.reshape(label, [CFG.num_classes])\n        return label\n    \n    def decode_with_labels(path, offset=None, label=None):\n        sig = decode_signal(path, offset)\n        label = decode_label(label)\n        return (sig, label)\n    \n    return decode_with_labels if with_labels else decode_signal","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n#前にあったやつ、ここにもってきた\n\ndecode_fn = build_decoder(labels is not None)\nds = ds.map(decode_fn, num_parallel_calls=AUTO)\n\n'''\nこのコードは、`decode_fn` 関数を使用してデータセット内の各要素をデコードし、新しいデータセットを作成します。\n`map()` 関数は、各要素に対して指定された関数を適用するために使用されます。\n\n具体的には、次の手順が行われます：\n\n1. `ds.map(decode_fn, num_parallel_calls=AUTO)`：\n`decode_fn` 関数を使用してデータセット内の各要素をデコードするための新しいデータセットを作成します。\n`num_parallel_calls=AUTO` パラメータは、デコード処理を並行して実行するためのオプションです。\n`AUTO` を指定することで、TensorFlowが自動的に最適な並列実行数を選択します。\n\n2. `decode_fn` 関数は、各要素をデコードするための処理を行います。\nこれには、例えば画像のデコードや正規化、テキストのトークン化やエンコードなどが含まれる場合があります。\nデコード処理は、データセット内の各要素に対して適用されます。\n\nしたがって、このコードは、与えられたデコード関数を使用してデータセット内の各要素をデコードし、その結果を新しいデータセットとして返します。\n\n'''","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    \n\n\n\n\n\n\n\n\n\nds = ds.cache(cache_dir) if cache else ds\nds = ds.repeat() if repeat else ds\nif shuffle: \n    ds = ds.shuffle(shuffle, seed=CFG.seed)\n    opt = tf.data.Options()\n    opt.experimental_deterministic = False\n    ds = ds.with_options(opt)\nds = ds.batch(batch_size, drop_remainder=drop_remainder)\nds = ds.map(augment_fn, num_parallel_calls=AUTO) if augment else ds\nds = ds.prefetch(AUTO)\nreturn ds\n\nprint('panda')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = build_dataset(train_paths, \n                         train_offsets, \n                         train_labels, \n                         batch_size=CFG.batch_size,\n                         repeat=True, \n                         shuffle=True, \n                         augment=True, \n                         cache=True)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_df = sample_df[sample_df.fold == CFG.fold]\nprint(f\"# Num Train: {len(train_df)} | Num Valid: {len(valid_df)}\")\n\n# Valid\nvalid_paths = valid_df.spec2_path.values\nvalid_offsets = valid_df.spectrogram_label_offset_seconds.values.astype(int)\nvalid_labels = valid_df.class_label.values\nvalid_ds = build_dataset(valid_paths, valid_offsets, valid_labels, batch_size=CFG.batch_size,\n                         repeat=False, shuffle=False, augment=False, cache=True)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(train_df))\ntrain_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds\ntrain_ds.numpy","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"type(train_ds)\n<_PrefetchDataset element_spec=(TensorSpec(shape=(None, 400, 300, 3), dtype=tf.float32, name=None), TensorSpec(shape=(None, 6), dtype=tf.float32, name=None))>\n\nIt looks like you've provided the signature of a TensorFlow dataset, specifically a <PrefetchDataset>. This dataset seems to yield batches of data, where each batch consists of two tensors:\n\nA tensor with shape (None, 400, 300, 3) of type float32, representing images with dimensions 400x300 and 3 color channels (presumably RGB).\nAnother tensor with shape (None, 6) of type float32, which could be labels or some other associated data with each image.\nThe None dimension indicates that the batch size can vary, likely depending on how many samples are loaded into each batch during training or evaluation.","metadata":{}},{"cell_type":"markdown","source":"Yes, you can convert TensorFlow tensors to PyTorch tensors without using a for loop by leveraging vectorized operations. You can convert the entire batch of tensors in one go. Here's how you can do it:\nIn this approach, dataset.batch(len(dataset)) batches the entire dataset into one batch. Then, next(iter(...)) extracts this batch. This approach assumes that your dataset can fit entirely into memory for batch conversion. If your dataset is too large to fit into memory all at once, you might need to iterate over it in smaller batches.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport torch\n\n# Assuming your dataset is named 'train_ds' and you want to convert the first 'num_samples' samples\nnum_samples = 1000  # Adjust as needed\nimages_tensor, labels_tensor = next(iter(train_ds.take(num_samples).batch(num_samples)))\n\n# Convert TensorFlow tensors to NumPy arrays\nimages_array = images_tensor.numpy()\nlabels_array = labels_tensor.numpy()\n\n# Convert NumPy arrays to PyTorch tensors\nimages_torch = torch.tensor(images_array)\nlabels_torch = torch.tensor(labels_array)\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchvision\nimport torchvision.transforms as transforms\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# Device configuration\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n# Hyper-parameters \nnum_epochs = 5\nbatch_size = 4\nlearning_rate = 0.001\n\n# dataset has PILImage images of range [0, 1]. \n# We transform them to Tensors of normalized range [-1, 1]\ntransform = transforms.Compose(\n    [transforms.ToTensor(),\n     transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))])\n\n# CIFAR10: 60000 32x32 color images in 10 classes, with 6000 images per class\ntrain_dataset = torchvision.datasets.CIFAR10(root='./data', train=True,\n                                        download=True, transform=transform)\n\ntest_dataset = torchvision.datasets.CIFAR10(root='./data', train=False,\n                                       download=True, transform=transform)\n\n\n\n\n\n\ntrain_loader = torch.utils.data.DataLoader(train_dataset, batch_size=batch_size,\n                                          shuffle=True)\n\ntest_loader = torch.utils.data.DataLoader(test_dataset, batch_size=batch_size,\n                                         shuffle=False)\n\nclasses = ('plane', 'car', 'bird', 'cat',\n           'deer', 'dog', 'frog', 'horse', 'ship', 'truck')\n\ndef imshow(img):\n    img = img / 2 + 0.5  # unnormalize\n    npimg = img.numpy()\n    plt.imshow(np.transpose(npimg, (1, 2, 0)))\n    plt.show()\n\n\n# get some random training images\ndataiter = iter(train_loader)\nimages, labels = next(dataiter)\n\n# show images\nimshow(torchvision.utils.make_grid(images))\n\nclass ConvNet(nn.Module):\n    def __init__(self):\n        super(ConvNet, self).__init__()\n        self.conv1 = nn.Conv2d(3, 6, 5)\n        self.pool = nn.MaxPool2d(2, 2)\n        self.conv2 = nn.Conv2d(6, 16, 5)\n        self.fc1 = nn.Linear(16 * 5 * 5, 120)\n        self.fc2 = nn.Linear(120, 84)\n        self.fc3 = nn.Linear(84, 10)\n\n    def forward(self, x):\n        # -> n, 3, 32, 32\n        x = self.pool(F.relu(self.conv1(x)))  # -> n, 6, 14, 14\n        x = self.pool(F.relu(self.conv2(x)))  # -> n, 16, 5, 5\n        x = x.view(-1, 16 * 5 * 5)            # -> n, 400\n        x = F.relu(self.fc1(x))               # -> n, 120\n        x = F.relu(self.fc2(x))               # -> n, 84\n        x = self.fc3(x)                       # -> n, 10\n        return x\n\n\nmodel = ConvNet().to(device)\n\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.SGD(model.parameters(), lr=learning_rate)\n\nn_total_steps = len(train_loader)\nfor epoch in range(num_epochs):\n    for i, (images, labels) in enumerate(train_loader):\n        # origin shape: [4, 3, 32, 32] = 4, 3, 1024\n        # input_layer: 3 input channels, 6 output channels, 5 kernel size\n        images = images.to(device)\n        labels = labels.to(device)\n\n        # Forward pass\n        outputs = model(images)\n        loss = criterion(outputs, labels)\n\n        # Backward and optimize\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        if (i+1) % 2000 == 0:\n            print (f'Epoch [{epoch+1}/{num_epochs}], Step [{i+1}/{n_total_steps}], Loss: {loss.item():.4f}')\n\nprint('Finished Training')\nPATH = './cnn.pth'\ntorch.save(model.state_dict(), PATH)\n\nwith torch.no_grad():\n    n_correct = 0\n    n_samples = 0\n    n_class_correct = [0 for i in range(10)]\n    n_class_samples = [0 for i in range(10)]\n    for images, labels in test_loader:\n        images = images.to(device)\n        labels = labels.to(device)\n        outputs = model(images)\n        # max returns (value ,index)\n        _, predicted = torch.max(outputs, 1)\n        n_samples += labels.size(0)\n        n_correct += (predicted == labels).sum().item()\n        \n        for i in range(batch_size):\n            label = labels[i]\n            pred = predicted[i]\n            if (label == pred):\n                n_class_correct[label] += 1\n            n_class_samples[label] += 1\n\n    acc = 100.0 * n_correct / n_samples\n    print(f'Accuracy of the network: {acc} %')\n\n    for i in range(10):\n        acc = 100.0 * n_class_correct[i] / n_class_samples[i]\n        print(f'Accuracy of {classes[i]}: {acc} %')\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming train_ds is a tuple of Python objects representing samples\n#train_ds = (([1, 2, 3],), ([4, 5, 6],))\n\n# Print the first sample and its shape\nfor sample in train_ds:\n    print(\"Values:\", sample[0]) #shape=(64, 400, 300, 3), dtype=float32)\n    print(\"Shape: (\", len(sample[0]), \")\")  #Shape: ( 64 )\n    break  # Print only the first sample\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset Check\n\nLet's visualize some samples from the dataset.","metadata":{}},{"cell_type":"code","source":"imgs, tars = next(iter(train_ds))\n\nnum_imgs = 8\nplt.figure(figsize=(4*4, num_imgs//4*5))\nfor i in range(num_imgs):\n    plt.subplot(num_imgs//4, 4, i + 1)\n    img = imgs[i].numpy()[...,0]  # Adjust as per your image data format\n    img -= img.min()\n    img /= img.max() + 1e-4\n    tar = CFG.label2name[np.argmax(tars[i].numpy())]\n    plt.imshow(img)\n    plt.title(f\"Target: {tar}\")\n    plt.axis('off')\n    \nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🔍 | Loss & Metric\n\nThe evaluation metric in this competition is **KL Divergence**, defined as,\n\n$$\nD_{\\text{KL}}(P \\parallel Q) = \\sum_{i} P(i) \\log\\left(\\frac{P(i)}{Q(i)}\\right)\n$$\n\nWhere:\n- $P$ is the true distribution.\n- $Q$ is the predicted distribution.\n\nInterestingly, as KL Divergence is differentiable, we can directly use it as our loss function. Thus, we don't need to use a third-party metric like **Accuracy** to evaluate our model. Therefore, `valid_loss` can stand alone as an indicator for our evaluation. In keras, we already have impelementation for KL Divergence loss so we only need to import it.","metadata":{}},{"cell_type":"code","source":"LOSS = keras.losses.KLDivergence()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🤖 | Modeling\n\nThis notebook uses the `EfficientNetV2 B2` from KerasCV's collection of pretrained models. To explore other models, simply modify the `preset` in the `CFG` (config). Check the [KerasCV website](https://keras.io/api/keras_cv/models/tasks/image_classifier/) for a list of available pretrained models.","metadata":{"papermill":{"duration":0.016849,"end_time":"2024-01-14T03:18:38.613991","exception":false,"start_time":"2024-01-14T03:18:38.597142","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Build Classifier\nmodel = keras_cv.models.ImageClassifier.from_preset(\n    CFG.preset, num_classes=CFG.num_classes\n)\n\n# Compile the model  \nmodel.compile(optimizer=keras.optimizers.Adam(learning_rate=1e-4),\n              loss=LOSS)\n\n# Model Sumamry\nmodel.summary()","metadata":{"papermill":{"duration":10.446166,"end_time":"2024-01-14T03:18:49.186176","exception":false,"start_time":"2024-01-14T03:18:38.74001","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ⚓ | LR Schedule\n\nA well-structured learning rate schedule is essential for efficient model training, ensuring optimal convergence and avoiding issues such as overshooting or stagnation.","metadata":{"papermill":{"duration":0.016209,"end_time":"2024-01-14T03:18:49.21924","exception":false,"start_time":"2024-01-14T03:18:49.203031","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import math\n\ndef get_lr_callback(batch_size=8, mode='cos', epochs=10, plot=False):\n    lr_start, lr_max, lr_min = 5e-5, 6e-6 * batch_size, 1e-5\n    lr_ramp_ep, lr_sus_ep, lr_decay = 3, 0, 0.75\n\n    def lrfn(epoch):  # Learning rate update function\n        if epoch < lr_ramp_ep: lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n        elif epoch < lr_ramp_ep + lr_sus_ep: lr = lr_max\n        elif mode == 'exp': lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n        elif mode == 'step': lr = lr_max * lr_decay**((epoch - lr_ramp_ep - lr_sus_ep) // 2)\n        elif mode == 'cos':\n            decay_total_epochs, decay_epoch_index = epochs - lr_ramp_ep - lr_sus_ep + 3, epoch - lr_ramp_ep - lr_sus_ep\n            phase = math.pi * decay_epoch_index / decay_total_epochs\n            lr = (lr_max - lr_min) * 0.5 * (1 + math.cos(phase)) + lr_min\n        return lr\n\n    if plot:  # Plot lr curve if plot is True\n        plt.figure(figsize=(10, 5))\n        plt.plot(np.arange(epochs), [lrfn(epoch) for epoch in np.arange(epochs)], marker='o')\n        plt.xlabel('epoch'); plt.ylabel('lr')\n        plt.title('LR Scheduler')\n        plt.show()\n\n    return keras.callbacks.LearningRateScheduler(lrfn, verbose=False)  # Create lr callback","metadata":{"papermill":{"duration":0.028945,"end_time":"2024-01-14T03:18:49.264535","exception":false,"start_time":"2024-01-14T03:18:49.23559","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_cb = get_lr_callback(CFG.batch_size, mode=CFG.lr_mode, plot=True)","metadata":{"papermill":{"duration":0.297147,"end_time":"2024-01-14T03:18:49.578089","exception":false,"start_time":"2024-01-14T03:18:49.280942","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 💾 | Model Checkpointing","metadata":{"papermill":{"duration":0.017199,"end_time":"2024-01-14T03:18:49.613648","exception":false,"start_time":"2024-01-14T03:18:49.596449","status":"completed"},"tags":[]}},{"cell_type":"code","source":"ckpt_cb = keras.callbacks.ModelCheckpoint(\"best_model.keras\",\n                                         monitor='val_loss',\n                                         save_best_only=True,\n                                         save_weights_only=False,\n                                         mode='min')","metadata":{"papermill":{"duration":0.024529,"end_time":"2024-01-14T03:18:49.655708","exception":false,"start_time":"2024-01-14T03:18:49.631179","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🚂 | Training","metadata":{"papermill":{"duration":0.01671,"end_time":"2024-01-14T03:18:49.689354","exception":false,"start_time":"2024-01-14T03:18:49.672644","status":"completed"},"tags":[]}},{"cell_type":"code","source":"history = model.fit(\n    train_ds, \n    epochs=CFG.epochs,\n    callbacks=[lr_cb, ckpt_cb], \n    steps_per_epoch=len(train_df)//CFG.batch_size,\n    validation_data=valid_ds, \n    verbose=CFG.verbose\n)","metadata":{"papermill":{"duration":3374.692199,"end_time":"2024-01-14T04:15:04.398389","exception":false,"start_time":"2024-01-14T03:18:49.70619","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🧪 | Prediction","metadata":{"papermill":{"duration":0.693309,"end_time":"2024-01-14T04:15:05.731839","exception":false,"start_time":"2024-01-14T04:15:05.03853","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Load Best Model","metadata":{"papermill":{"duration":0.632183,"end_time":"2024-01-14T04:15:06.991143","exception":false,"start_time":"2024-01-14T04:15:06.35896","status":"completed"},"tags":[]}},{"cell_type":"code","source":"model.load_weights(\"best_model.keras\")","metadata":{"papermill":{"duration":20.428261,"end_time":"2024-01-14T04:15:28.044401","exception":false,"start_time":"2024-01-14T04:15:07.61614","status":"completed"},"tags":[],"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build Test Dataset","metadata":{"papermill":{"duration":0.703901,"end_time":"2024-01-14T04:20:09.745279","exception":false,"start_time":"2024-01-14T04:20:09.041378","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test_paths = test_df.spec2_path.values\ntest_ds = build_dataset(test_paths, batch_size=min(CFG.batch_size, len(test_df)),\n                         repeat=False, shuffle=False, cache=False, augment=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"preds = model.predict(test_ds)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📩 | Submission","metadata":{}},{"cell_type":"code","source":"pred_df = test_df[[\"eeg_id\"]].copy()\ntarget_cols = [x.lower()+'_vote' for x in CFG.class_names]\npred_df[target_cols] = preds.tolist()\n\nsub_df = pd.read_csv(f'{BASE_PATH}/sample_submission.csv')\nsub_df = sub_df[[\"eeg_id\"]].copy()\nsub_df = sub_df.merge(pred_df, on=\"eeg_id\", how=\"left\")\nsub_df.to_csv(\"submission.csv\", index=False)\nsub_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📌 | Reference\n* [HMS-HBAC: ResNet34d Baseline [Training]](https://www.kaggle.com/code/ttahara/hms-hbac-resnet34d-baseline-training) \n* [EfficientNetB2 Starter - [LB 0.57]](https://www.kaggle.com/code/cdeotte/efficientnetb2-starter-lb-0-57)","metadata":{}}]}