{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":37174,"databundleVersionId":3938797,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"f3825ed7","cell_type":"code","source":"!pip install transformers torch librosa pandas -q","metadata":{},"outputs":[],"execution_count":null},{"id":"51b5df5e","cell_type":"code","source":"import torch\nimport librosa\nimport pandas as pd\nimport os\nfrom pathlib import Path\nfrom transformers import Wav2Vec2ForSequenceClassification, Wav2Vec2FeatureExtractor\nimport numpy as np\nimport warnings\nfrom tqdm import tqdm\nwarnings.filterwarnings('ignore')\n\n# Configuration\nBASE_PATH = \"/kaggle/input/dlsprint\"\nAUDIO_FOLDER = f\"{BASE_PATH}/train_files\"\nCSV_FILE = f\"{BASE_PATH}/train.csv\"\nOUTPUT_CSV = \"dlsprint_emotion_classified.csv\"\n\n# Batch processing configuration\nBATCH_SIZE = 64  # Adjust based on GPU memory\nDEVICE = \"cuda:0\" if torch.cuda.is_available() else \"cpu\"","metadata":{},"outputs":[],"execution_count":null},{"id":"80e4c3a3","cell_type":"code","source":"# Load and prepare data\nprint(\"Loading train.csv...\")\ndf = pd.read_csv(CSV_FILE)\n\nprint(f\"Total rows in CSV: {len(df)}\")\nprint(f\"Columns: {df.columns.tolist()}\")\n\n# Filter for \"শুদ্ধ বাংলা\" accent only\ndf_filtered = df[df['accents'] == 'শুদ্ধ বাংলা'].copy()\nprint(f\"\\nRows after filtering for 'শুদ্ধ বাংলা': {len(df_filtered)}\")\n\n# Select only required columns\ndf_filtered = df_filtered[['path', 'gender', 'sentence']].copy()\n\n# Add full audio path\ndf_filtered['audio_path'] = df_filtered['path'].apply(lambda x: f\"{AUDIO_FOLDER}/{x}\")\n\n# Verify files exist\ndf_filtered['file_exists'] = df_filtered['audio_path'].apply(os.path.exists)\nmissing_files = (~df_filtered['file_exists']).sum()\nprint(f\"Missing audio files: {missing_files}\")\n\n# Keep only existing files\ndf_filtered = df_filtered[df_filtered['file_exists']].copy()\ndf_filtered = df_filtered.drop('file_exists', axis=1)\n\nprint(f\"\\nFinal dataset size: {len(df_filtered)}\")\nprint(f\"\\nGender distribution:\")\nprint(df_filtered['gender'].value_counts())\nprint(f\"\\nSample data:\")\nprint(df_filtered.head())","metadata":{},"outputs":[],"execution_count":null},{"id":"75c5cb22","cell_type":"code","source":"def load_audio_batch(audio_paths, target_sr=16000):\n    \"\"\"Load multiple audio files and return them as a batch\"\"\"\n    audio_data = []\n    valid_indices = []\n    \n    for idx, path in enumerate(audio_paths):\n        try:\n            speech, sr = librosa.load(path, sr=target_sr)\n            audio_data.append(speech)\n            valid_indices.append(idx)\n        except Exception as e:\n            print(f\"Error loading {path}: {e}\")\n    \n    return audio_data, valid_indices\n\n\ndef classify_batch(audio_batch, model, feature_extractor, device):\n    \"\"\"Classify a batch of audio files and return emotions with probabilities\"\"\"\n    if not audio_batch:\n        return []\n    \n    try:\n        # Process batch\n        inputs = feature_extractor(\n            audio_batch, \n            sampling_rate=16000, \n            return_tensors=\"pt\", \n            padding=True\n        )\n        \n        # Move to device\n        inputs = {k: v.to(device) for k, v in inputs.items()}\n        \n        with torch.no_grad():\n            logits = model(**inputs).logits\n        \n        # Get probabilities\n        probs = torch.nn.functional.softmax(logits, dim=-1)\n        predicted_ids = torch.argmax(probs, dim=-1)\n        \n        results = []\n        labels = model.config.id2label\n        \n        for i in range(len(predicted_ids)):\n            emotion = labels[predicted_ids[i].item()]\n            confidence = probs[i][predicted_ids[i]].item()\n            results.append((emotion, confidence))\n        \n        return results\n    except Exception as e:\n        print(f\"Error in batch classification: {e}\")\n        return [(None, 0.0)] * len(audio_batch)","metadata":{},"outputs":[],"execution_count":null},{"id":"08d4cdb8","cell_type":"code","source":"def process_audio_files_batch(audio_paths, model, feature_extractor, device, batch_size=64):\n    \"\"\"Process all audio files in batches for efficient GPU utilization\"\"\"\n    all_emotions = []\n    all_confidences = []\n    \n    print(f\"Processing {len(audio_paths)} audio files in batches of {batch_size}...\")\n    \n    for i in tqdm(range(0, len(audio_paths), batch_size), desc=\"Processing batches\"):\n        batch_paths = audio_paths[i:i + batch_size]\n        \n        # Load audio batch\n        audio_batch, valid_indices = load_audio_batch(batch_paths)\n        \n        if audio_batch:\n            # Classify batch\n            batch_results = classify_batch(audio_batch, model, feature_extractor, device)\n            \n            # Map results back\n            result_idx = 0\n            for local_idx in range(len(batch_paths)):\n                if local_idx in valid_indices:\n                    emotion, confidence = batch_results[result_idx]\n                    all_emotions.append(emotion)\n                    all_confidences.append(confidence)\n                    result_idx += 1\n                else:\n                    all_emotions.append(None)\n                    all_confidences.append(0.0)\n        else:\n            # No valid audio in this batch\n            all_emotions.extend([None] * len(batch_paths))\n            all_confidences.extend([0.0] * len(batch_paths))\n    \n    return all_emotions, all_confidences","metadata":{},"outputs":[],"execution_count":null},{"id":"49c31413","cell_type":"code","source":"# Load model\nprint(\"Loading emotion classification model...\")\nmodel_name = \"ehcalabres/wav2vec2-lg-xlsr-en-speech-emotion-recognition\"\nfeature_extractor = Wav2Vec2FeatureExtractor.from_pretrained(model_name)\nmodel = Wav2Vec2ForSequenceClassification.from_pretrained(model_name)\nmodel = model.to(DEVICE)\nmodel.eval()\nprint(f\"Model loaded on {DEVICE}\")","metadata":{},"outputs":[],"execution_count":null},{"id":"cb3ff1c3","cell_type":"code","source":"# Check GPU availability and memory\nprint(\"\\nChecking GPU availability...\")\nprint(f\"CUDA available: {torch.cuda.is_available()}\")\nif torch.cuda.is_available():\n    print(f\"GPU: {torch.cuda.get_device_name(0)}\")\n    print(f\"Total GPU memory: {torch.cuda.get_device_properties(0).total_memory / 1024**3:.2f} GB\")\n    print(f\"Current batch size: {BATCH_SIZE}\")\n    print(f\"Allocated memory: {torch.cuda.memory_allocated(0) / 1024**3:.2f} GB\")\nelse:\n    print(\"WARNING: No GPU available! Processing will be slow.\")","metadata":{},"outputs":[],"execution_count":null},{"id":"102eab41","cell_type":"code","source":"# Main processing\nprint(\"\\n\" + \"=\"*60)\nprint(\"Starting emotion classification...\")\nprint(\"=\"*60)\n\naudio_paths = df_filtered['audio_path'].tolist()\n\n# Process all audio files\nemotions, confidences = process_audio_files_batch(\n    audio_paths, \n    model, \n    feature_extractor, \n    DEVICE, \n    batch_size=BATCH_SIZE\n)\n\nprint(\"\\nClassification completed!\")","metadata":{},"outputs":[],"execution_count":null},{"id":"b172ff31","cell_type":"code","source":"# Create final dataset\nprint(\"\\nCreating final dataset...\")\ndf_filtered['emotion_class'] = emotions\ndf_filtered['confidence'] = confidences\n\n# Remove rows where emotion classification failed\ndf_final = df_filtered[df_filtered['emotion_class'].notna()].copy()\n\n# Drop temporary columns\ndf_final = df_final[['path', 'gender', 'sentence', 'emotion_class']].copy()\n\nprint(f\"Successfully classified: {len(df_final)} / {len(df_filtered)} audio files\")\n\n# Display emotion distribution\nprint(\"\\nEmotion distribution:\")\nprint(df_final['emotion_class'].value_counts())\n\n# Emotion by gender\nprint(\"\\nEmotion distribution by gender:\")\nprint(pd.crosstab(df_final['gender'], df_final['emotion_class']))\n\n# Display sample\nprint(\"\\nSample data:\")\nprint(df_final.head(10))","metadata":{},"outputs":[],"execution_count":null},{"id":"82c4d7bf","cell_type":"code","source":"# Save to CSV\ndf_final.to_csv(OUTPUT_CSV, index=False, encoding='utf-8')\nprint(f\"\\n✅ Dataset saved to: {OUTPUT_CSV}\")\nprint(f\"Columns: {list(df_final.columns)}\")\nprint(f\"Total rows: {len(df_final)}\")\nprint(f\"\\nFile size: {os.path.getsize(OUTPUT_CSV) / 1024:.2f} KB\")","metadata":{},"outputs":[],"execution_count":null}]}