{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":129276,"databundleVersionId":15506988,"sourceType":"competition"},{"sourceId":14804321,"sourceType":"datasetVersion","datasetId":9466139}],"dockerImageVersionId":31260,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-11T03:14:48.317216Z","iopub.execute_input":"2026-02-11T03:14:48.317531Z","iopub.status.idle":"2026-02-11T03:14:50.081343Z","shell.execute_reply.started":"2026-02-11T03:14:48.317504Z","shell.execute_reply":"2026-02-11T03:14:50.079938Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install silero-vad","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-11T03:49:14.611634Z","iopub.execute_input":"2026-02-11T03:49:14.612356Z","iopub.status.idle":"2026-02-11T03:52:35.384863Z","shell.execute_reply.started":"2026-02-11T03:49:14.612322Z","shell.execute_reply":"2026-02-11T03:52:35.383863Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport subprocess\n\n# This searches for any folder named 'assets' in your inputs\npossible_path = \"\"\nfor root, dirs, files in os.walk('/kaggle/input'):\n    if 'assets' in dirs:\n        possible_path = os.path.join(root, 'assets')\n        break\n\nif possible_path:\n    print(f\"Found assets at: {possible_path}\")\n    lib_path = possible_path\n    \n    # Manual install loop\n    whl_files = [f for f in os.listdir(lib_path) if f.endswith('.whl')]\n    for file in whl_files:\n        full_path = os.path.join(lib_path, file)\n        # --no-deps is faster; --no-index keeps it offline for the rules\n        subprocess.run([\"pip\", \"install\", full_path, \"--no-index\", \"--no-deps\"])\n    print(\"Installation Complete!\")\nelse:\n    print(\"Could not find 'assets' folder. Please check if the dataset is attached.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-11T07:41:34.365828Z","iopub.execute_input":"2026-02-11T07:41:34.366385Z","iopub.status.idle":"2026-02-11T07:46:29.698552Z","shell.execute_reply.started":"2026-02-11T07:41:34.366357Z","shell.execute_reply":"2026-02-11T07:46:29.697686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\n\n# Save your current configuration so you don't have to find paths again\nconfig = {\n    'asset_path': '/kaggle/input/datasets/smsadikuzzamanabir/dl-sprint-v1/assets',\n    'model_name': 'large-v3-turbo',\n    'status': 'Environment Ready'\n}\n\njoblib.dump(config, '/kaggle/working/competition_config.pkl')\nprint(\"Progress saved. You can safely turn off the GPU now.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-11T08:41:40.34093Z","iopub.execute_input":"2026-02-11T08:41:40.341707Z","iopub.status.idle":"2026-02-11T08:41:40.37828Z","shell.execute_reply.started":"2026-02-11T08:41:40.341674Z","shell.execute_reply":"2026-02-11T08:41:40.377723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport subprocess\nimport joblib\nfrom bnunicodenormalizer import Normalizer\nfrom faster_whisper import WhisperModel\n\n# 1. Restore Paths from your saved config\ntry:\n    config = joblib.load('/kaggle/working/competition_config.pkl')\n    lib_path = config['asset_path']\n    print(f\"Welcome back! Assets found at: {lib_path}\")\nexcept:\n    # Fallback to the path we discovered earlier\n    lib_path = '/kaggle/input/datasets/smsadikuzzamanabir/dl-sprint-v1/assets'\n\n# 2. Re-install everything offline (Required every time the session restarts)\nwhl_files = [f for f in os.listdir(lib_path) if f.endswith('.whl')]\nfor file in whl_files:\n    subprocess.run([\"pip\", \"install\", os.path.join(lib_path, file), \"--no-index\", \"--no-deps\"])\n\n# 3. Re-initialize the Model and Normalizer\nbnorm = Normalizer()\nmodel = WhisperModel(\"large-v3-turbo\", device=\"cuda\", compute_type=\"float16\")\n\nprint(\"Environment restored. Ready to transcribe!\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}