{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaL4","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"},{"sourceId":91496,"databundleVersionId":11483707,"sourceType":"competition"},{"sourceId":11208315,"sourceType":"datasetVersion","datasetId":6998616}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%%writefile config.py\nfrom dataclasses import dataclass\nimport torch\nimport os\n\n@dataclass\nclass Config:\n    # Model configuration\n    model_name: str = \"Qwen/Qwen2.5-Coder-0.5B-Instruct\"\n    output_dir: str = \"./qwen2_mc_model\"\n    \n    # Training configuration\n    num_folds: int = 5\n    batch_size: int = 4  # Base batch size per GPU\n    num_epochs: int = 10\n    learning_rate: float = 1e-5\n    max_length: int = 1536\n    \n    # Optimizer settings\n    weight_decay: float = 0.01  # Added weight decay parameter\n    \n    # Scheduler settings\n    lr_scheduler_type: str = \"cosine\"  # Options: \"linear\", \"cosine\", \"constant\", etc.\n    num_warmup_steps: int = 0  # 0 means auto-calculate as 10% of training steps\n    \n    # Data configuration\n    data_path: str = \"/kaggle/input/fpt-ai-res-entry-test/processed_data.csv\"\n    \n    # Special modes\n    debug: bool = False\n    verbose: bool = False\n        \n    # LoRA configuration\n    use_lora: bool = False  # Whether to use LoRA\n    lora_r: int = 4  # LoRA rank\n    lora_alpha: int = 8  # LoRA alpha\n    lora_dropout: float = 0.05  # LoRA dropout\n    lora_target_modules = [\"q_proj\", \"k_proj\", \"v_proj\"]  # Will be set in __init__ based on model\n    \n    # Add special token settings (following DeBERTa approach)\n    special_tokens = {\n        'cls_a': '<CLS_A>',\n        'cls_b': '<CLS_B>',\n        'cls_c': '<CLS_C>',\n        'cls_d': '<CLS_D>',\n        'cls_e': '<CLS_E>'\n    }\n    fail_on_truncation = False  # Stop training if truncation of special tokens is detected\n    num_choices = 5  # Fixed number of choices\n    \n    def __init__(self):\n        # Device setup - exactly as in DeBERTa\n        self.device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n        self.num_gpus = torch.cuda.device_count() if torch.cuda.is_available() else 0\n        \n        if self.debug:\n            print(\"--- Running in DEBUG mode ---\")\n            self.num_epochs = 1\n            self.debug_limit = 100\n            self.verbose = True\n        else:\n            self.debug_limit = None\n            \n        os.makedirs(self.output_dir, exist_ok=True)\n        \n        if self.num_gpus > 1:\n            print(f\"Using {self.num_gpus} GPUs with DataParallel\")\n            self.effective_batch_size = self.batch_size * self.num_gpus\n        elif self.num_gpus == 1:\n            print(\"Using single GPU\")\n            self.effective_batch_size = self.batch_size\n        else:\n            print(\"Using CPU\")\n            self.effective_batch_size = self.batch_size\n            \n        if self.verbose:\n            print(\"--- Verbose mode enabled ---\")\n\n        if self.use_lora:\n            print(f\"LoRA enabled: r={self.lora_r}, alpha={self.lora_alpha}, targets={self.lora_target_modules}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-30T16:45:41.172036Z","iopub.execute_input":"2025-03-30T16:45:41.172256Z","iopub.status.idle":"2025-03-30T16:45:41.177315Z","shell.execute_reply.started":"2025-03-30T16:45:41.172235Z","shell.execute_reply":"2025-03-30T16:45:41.176654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile lora_utils.py\nfrom peft import (\n    get_peft_model,\n    LoraConfig,\n    TaskType,\n    prepare_model_for_kbit_training\n)\nimport torch\n\ndef apply_lora_to_model(model, config):\n    \"\"\"\n    Apply LoRA to a model and return the wrapped model.\n    \n    Args:\n        model: The base model to apply LoRA to\n        config: Configuration object with LoRA parameters\n    \"\"\"\n    print(f\"Applying LoRA with rank={config.lora_r}, alpha={config.lora_alpha}, dropout={config.lora_dropout}\")\n    \n    # Define the LoRA configuration\n    peft_config = LoraConfig(\n        task_type=None,  # Use None to avoid task-specific adaptations that might cause errors\n        inference_mode=False,\n        r=config.lora_r,\n        lora_alpha=config.lora_alpha,\n        lora_dropout=config.lora_dropout,\n        target_modules=config.lora_target_modules\n    )\n    \n    # Apply LoRA to the model\n    lora_model = get_peft_model(model, peft_config)\n    \n    # Ensure the classification head remains trainable after applying LoRA\n    if hasattr(lora_model, 'head'):\n        print(\"Setting head parameters to be trainable\")\n        for param in lora_model.head.parameters():\n            param.requires_grad = True\n    \n    # Print trainable parameters info\n    trainable_params = 0\n    all_param = 0\n    for _, param in lora_model.named_parameters():\n        all_param += param.numel()\n        if param.requires_grad:\n            trainable_params += param.numel()\n    \n    print(\n        f\"Trainable params: {trainable_params:,d} ({100 * trainable_params / all_param:.2f}%)\"\n        f\" of total {all_param:,d}\"\n    )\n    \n    return lora_model\n\ndef save_lora_weights(model, path):\n    \"\"\"Save just the LoRA weights from a PEFT model\"\"\"\n    if hasattr(model, \"save_pretrained\"):\n        model.save_pretrained(path)\n    else:\n        # Handle DataParallel wrapped models\n        if hasattr(model, \"module\") and hasattr(model.module, \"save_pretrained\"):\n            model.module.save_pretrained(path)\n        else:\n            raise ValueError(\"Model doesn't have save_pretrained method\")","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-03-30T16:45:41.178174Z","iopub.execute_input":"2025-03-30T16:45:41.178544Z","iopub.status.idle":"2025-03-30T16:45:41.19321Z","shell.execute_reply.started":"2025-03-30T16:45:41.178523Z","shell.execute_reply":"2025-03-30T16:45:41.192608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile data.py\nimport pandas as pd\nimport numpy as np\nimport torch\nfrom dataclasses import dataclass\nfrom typing import List\nfrom torch.utils.data import Dataset\n\n@dataclass\nclass MultipleChoiceData:\n    task_id: str\n    question: str\n    choices: List[str]\n    answer: str\n\ndef load_and_preprocess_data(file_path: str):\n    df = pd.read_csv(file_path)\n\n    # Replace NaN with empty string in choice columns\n    choice_cols = ['choice_A', 'choice_B', 'choice_C', 'choice_D', 'choice_E']\n    for col in choice_cols:\n        df[col] = df[col].fillna('')\n\n    # Convert to list of MultipleChoiceData objects\n    data = []\n    for _, row in df.iterrows():\n        choices = [row['choice_A'], row['choice_B'], row['choice_C'],\n                  row['choice_D'], row['choice_E']]\n        data.append(MultipleChoiceData(\n            task_id=row['task_id'],\n            question=row['question'],\n            choices=choices,\n            answer=row['answer']\n        ))\n    return data\n\ndef add_special_tokens_to_tokenizer(tokenizer, special_tokens):\n    \"\"\"Add custom special tokens to tokenizer and resize embeddings\"\"\"\n    tokens_to_add = list(special_tokens.values())\n    special_tokens_dict = {'additional_special_tokens': tokens_to_add}\n    num_added = tokenizer.add_special_tokens(special_tokens_dict)\n    print(f\"Added {num_added} special tokens to tokenizer vocabulary\")\n    return tokenizer\n\nclass MultipleChoiceDataset(torch.utils.data.Dataset):\n    def __init__(self, data: List[MultipleChoiceData], tokenizer, max_length: int = 512, config=None):\n        self.data = data\n        self.tokenizer = tokenizer\n        self.max_length = max_length\n        self.label_map = {'A': 0, 'B': 1, 'C': 2, 'D': 3, 'E': 4}\n        self.config = config\n\n        print(\"Max length set to:\", self.max_length)\n\n        # Add special tokens if config is provided\n        if config and hasattr(config, 'special_tokens'):\n            # Check if special tokens have been added already\n            example_token = config.special_tokens['cls_a']\n            if example_token not in tokenizer.get_vocab():\n                self.tokenizer = add_special_tokens_to_tokenizer(tokenizer, config.special_tokens)\n                print(f\"Added special choice tokens to tokenizer: {list(config.special_tokens.values())}\")\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        item = self.data[idx]\n\n        # Get special tokens if available\n        use_special_tokens = self.config and hasattr(self.config, 'special_tokens')\n        \n        if use_special_tokens:\n            # Get SEP token\n            sep = self.tokenizer.sep_token or self.tokenizer.eos_token\n            \n            # Get special CLS tokens for each choice\n            cls_tokens = []\n            for choice_key in ['cls_a', 'cls_b', 'cls_c', 'cls_d', 'cls_e']:\n                if choice_key in self.config.special_tokens:\n                    cls_tokens.append(self.config.special_tokens[choice_key])\n                else:\n                    cls_tokens.append(f\"<{choice_key.upper()}>\")\n            \n            # Create input with special tokens AFTER each choice (important for decoder-only models)\n            input_parts = [item.question, sep]\n            \n            # Track valid choices\n            valid_choice_mask = []\n            \n            # Add all choices with their special tokens AFTER the choice text\n            for i in range(self.config.num_choices):\n                cls_token = cls_tokens[i]\n                \n                # Get choice text or empty string if missing\n                if i < len(item.choices) and item.choices[i].strip():\n                    choice_text = item.choices[i]\n                    valid_choice_mask.append(1)\n                else:\n                    choice_text = \"\"\n                    valid_choice_mask.append(0)\n                \n                # Add the choice + special token + sep (token order matters for decoder models)\n                input_parts.append(choice_text)\n                input_parts.append(cls_token)\n                input_parts.append(sep)\n            \n            input_text = \"\".join(input_parts)\n            \n            # Tokenize the formatted text\n            encoding = self.tokenizer(\n                input_text,\n                max_length=self.max_length,\n                padding='max_length',\n                truncation=True,\n                return_tensors='pt'\n            )\n            \n            # Find positions of all CLS tokens\n            cls_positions = []\n            for cls_token in cls_tokens:\n                token_id = self.tokenizer.convert_tokens_to_ids(cls_token)\n                positions = [i for i, tid in enumerate(encoding['input_ids'].squeeze()) if tid == token_id]\n                # Take only the first occurrence if multiple (shouldn't happen)\n                if positions:\n                    cls_positions.append(positions[0])\n                else:\n                    # If token not found (likely due to truncation), add padding position 0\n                    cls_positions.append(0)\n            \n            # Remove batch dimension from tokenizer output\n            encoding = {k: v.squeeze(0) for k, v in encoding.items()}\n            encoding['labels'] = torch.tensor(self.label_map[item.answer]) if item.answer else torch.tensor(-1)\n            encoding['cls_positions'] = torch.tensor(cls_positions)\n            encoding['valid_choice_mask'] = torch.tensor(valid_choice_mask)\n            \n            # Add example ID for debugging\n            encoding['example_ids'] = item.task_id\n        else:\n            # Original implementation without special tokens\n            questions = [item.question] * 5\n            choices = item.choices\n\n            # Tokenize\n            encoding = self.tokenizer(\n                questions,\n                choices,\n                max_length=self.max_length,\n                padding='max_length',\n                truncation=True,\n                return_tensors='pt'\n            )\n\n            # Remove batch dimension from tokenizer output\n            encoding = {k: v.squeeze(0) for k, v in encoding.items()}\n            encoding['labels'] = torch.tensor(self.label_map[item.answer]) if item.answer else torch.tensor(-1)\n\n        return encoding\n\nclass CustomDataCollator:\n    \"\"\"Custom data collator that handles special fields like cls_positions and valid_choice_mask.\"\"\"\n    def __init__(self, tokenizer):\n        self.tokenizer = tokenizer\n        \n    def __call__(self, features):\n        # Extract special fields that need custom handling\n        cls_positions_list = []\n        valid_choice_masks_list = []\n        example_ids = []\n        \n        # First, handle special fields by removing them from features\n        for i, f in enumerate(features):\n            if \"cls_positions\" in f:\n                cls_positions = f.pop(\"cls_positions\")\n                cls_positions_list.append(cls_positions)\n                \n            if \"valid_choice_mask\" in f:\n                valid_choice_mask = f.pop(\"valid_choice_mask\") \n                valid_choice_masks_list.append(valid_choice_mask)\n                \n            if \"example_ids\" in f:\n                example_ids.append(f.pop(\"example_ids\"))\n        \n        # Handle standard fields\n        batch = {\n            k: torch.stack([f[k] for f in features]) \n            for k in features[0].keys()\n        }\n        \n        # Now add back our special fields with proper tensor conversion\n        if cls_positions_list:\n            batch[\"cls_positions\"] = torch.stack(cls_positions_list)\n            \n        if valid_choice_masks_list:\n            batch[\"valid_choice_mask\"] = torch.stack(valid_choice_masks_list)\n            \n        # Add example_ids back if we have them\n        if example_ids:\n            batch[\"example_ids\"] = example_ids\n            \n        return batch\n","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-03-30T16:45:41.382157Z","iopub.execute_input":"2025-03-30T16:45:41.382441Z","iopub.status.idle":"2025-03-30T16:45:41.387296Z","shell.execute_reply.started":"2025-03-30T16:45:41.382419Z","shell.execute_reply":"2025-03-30T16:45:41.386691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile model.py\nimport torch\nfrom torch import nn\nfrom transformers import AutoModel\nfrom transformers.modeling_outputs import MultipleChoiceModelOutput\nimport numpy as np\n\nclass ChoiceCLSTokensHead(nn.Module):\n    \"\"\"Head that uses dedicated classification tokens for each choice position.\"\"\"\n    \n    def __init__(self, hidden_size, num_choices=5):\n        super().__init__()\n        self.dropout = nn.Dropout(0.1)\n        self.classifier = nn.Linear(hidden_size, 1)\n        self.num_choices = num_choices\n        # For tracking truncation problems\n        self.truncation_detected = False\n        self.truncated_examples = []\n    \n    def forward(self, sequence_output, cls_positions=None, valid_choice_mask=None, example_ids=None):\n        \"\"\"\n        Forward pass that extracts representations at dedicated CLS token positions\n        \n        Args:\n            sequence_output: Output sequence from encoder\n            cls_positions: Positions of special CLS tokens for each choice\n            valid_choice_mask: Mask indicating which choices are valid\n            example_ids: Optional batch IDs for debugging\n        \"\"\"\n        device = sequence_output.device\n        batch_size = sequence_output.size(0)\n        \n        # Make sure we're not exceeding batch dimensions\n        safe_batch_size = batch_size\n        if cls_positions is not None:\n            safe_batch_size = min(batch_size, len(cls_positions))\n        if valid_choice_mask is not None:\n            safe_batch_size = min(safe_batch_size, len(valid_choice_mask))\n        \n        # Create default example_ids if not provided\n        if example_ids is None:\n            example_ids = [f\"idx_{i}\" for i in range(safe_batch_size)]\n        \n        # Initialize logits tensor with negative infinity\n        logits = torch.full((batch_size, self.num_choices), float('-inf'), device=device)\n        \n        # Check if we're missing cls_positions\n        if cls_positions is None:\n            raise ValueError(\"cls_positions cannot be None for ChoiceCLSTokensHead\")\n        \n        # Process each example in the batch\n        for i in range(safe_batch_size):\n            current_id = example_ids[i] if i < len(example_ids) else f\"idx_{i}\"\n            \n            # Get the CLS positions for this example\n            if i >= len(cls_positions):\n                print(f\"Warning: Missing CLS positions for example {current_id}\")\n                continue\n                \n            positions = cls_positions[i]\n            \n            # Check number of CLS tokens matches expected\n            expected_cls = self.num_choices  # One for each choice position\n            if len(positions) != expected_cls:\n                missing = expected_cls - len(positions)\n                print(f\"Warning: Example {current_id} is missing {missing} CLS tokens\")\n                \n                # Record truncation detection\n                self.truncation_detected = True\n                self.truncated_examples.append(current_id)\n            \n            # Get token representations and compute scores for each choice\n            choice_reps = []\n            valid_indices = []\n            \n            # Get the choice representations and corresponding valid indices\n            for j, pos in enumerate(positions):\n                if pos > 0:  # valid position (0 is used for padding)\n                    # Check if position is in bounds\n                    if pos >= sequence_output.shape[1]:\n                        print(f\"Warning: CLS position {pos} out of bounds for example {current_id}\")\n                        continue\n                    \n                    # Get token representation\n                    token_rep = sequence_output[i, pos]\n                    choice_reps.append(token_rep)\n                    valid_indices.append(j)\n            \n            # Skip if no valid representations found\n            if not choice_reps:\n                print(f\"Example {current_id}: No valid CLS token positions found\")\n                continue\n                \n            # Stack representations and apply dropout\n            stacked_reps = torch.stack(choice_reps)\n            stacked_reps = self.dropout(stacked_reps)\n            \n            # Generate scores for each choice\n            choice_scores = self.classifier(stacked_reps).squeeze(-1)\n            \n            # Apply scores to the appropriate positions in the logits\n            if valid_choice_mask is not None and i < len(valid_choice_mask):\n                # Check valid_choice_mask for this example\n                mask = valid_choice_mask[i]\n                \n                # Map scores to their positions in the logits tensor\n                # Only using valid indices that we found representations for\n                for score_idx, choice_idx in enumerate(valid_indices):\n                    if score_idx < len(choice_scores):\n                        logits[i, choice_idx] = choice_scores[score_idx]\n            else:\n                # Without a valid_choice_mask, assign scores to all positions we found\n                for idx, choice_idx in enumerate(valid_indices):\n                    if idx < len(choice_scores):\n                        logits[i, choice_idx] = choice_scores[idx]\n        \n        return logits\n\nclass Qwen2ForMultipleChoice(nn.Module):\n    def __init__(self, model_name, num_choices=5):\n        super().__init__()\n        self.model = AutoModel.from_pretrained(model_name)\n        self.num_choices = num_choices\n        # Replace simple classifier with the ChoiceCLSTokensHead\n        self.head = ChoiceCLSTokensHead(self.model.config.hidden_size, num_choices)\n        \n        # Add config attribute required by PEFT\n        if not hasattr(self, 'config') and hasattr(self.model, 'config'):\n            self.config = self.model.config\n\n    def forward(\n        self,\n        input_ids=None,\n        attention_mask=None,\n        token_type_ids=None,\n        position_ids=None,\n        labels=None,\n        cls_positions=None,\n        valid_choice_mask=None,\n        example_ids=None,\n        **kwargs\n    ):\n        # Standard base model forward pass\n        outputs = self.model(\n            input_ids=input_ids,\n            attention_mask=attention_mask,\n            # Add any other important parameters the base model needs\n            return_dict=True\n        )\n        \n        # Get the sequence output\n        sequence_output = outputs.last_hidden_state\n        \n        # Pass through the choice CLS tokens head\n        logits = self.head(sequence_output, cls_positions, valid_choice_mask, example_ids)\n\n        loss = None\n        if labels is not None:\n            loss_fct = nn.CrossEntropyLoss()\n            loss = loss_fct(logits, labels)\n\n        return MultipleChoiceModelOutput(\n            loss=loss,\n            logits=logits,\n            hidden_states=outputs.hidden_states,\n            attentions=outputs.attentions,\n        )\n\n    # Add a helper method to satisfy PEFT requirements\n    def get_input_embeddings(self):\n        \"\"\"Returns input embeddings from the base model\"\"\"\n        return self.model.get_input_embeddings()\n\n    def get_output_embeddings(self):\n        \"\"\"Returns None since we don't use language modeling head\"\"\"\n        return None\n\ndef compute_metrics(eval_pred):\n    predictions, labels = eval_pred\n    preds = np.argmax(predictions, axis=1)\n    accuracy = (preds == labels).mean()\n    return {\"accuracy\": accuracy}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-30T16:45:41.388136Z","iopub.execute_input":"2025-03-30T16:45:41.38836Z","iopub.status.idle":"2025-03-30T16:45:41.402444Z","shell.execute_reply.started":"2025-03-30T16:45:41.388341Z","shell.execute_reply":"2025-03-30T16:45:41.401859Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile multi_gpu.py\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import KFold\nfrom transformers import AutoTokenizer, AutoModel, TrainingArguments\nimport torch\nfrom torch import nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader\nfrom transformers.modeling_outputs import MultipleChoiceModelOutput\nfrom typing import Dict, List, Optional\nimport os\nfrom dataclasses import dataclass\nfrom tqdm.auto import tqdm\n# Add PEFT imports for LoRA (though not fully used in the current setup)\n# from peft import LoraConfig, get_peft_model, PeftModel, PeftConfig, TaskType\n\n# Data class to structure the multiple choice input\n@dataclass\nclass MultipleChoiceData:\n    task_id: str\n    question: str\n    choices: List[str]\n    answer: str\n\n# Function to load and preprocess data\ndef load_and_preprocess_data(file_path: str):\n    df = pd.read_csv(file_path)\n\n    # Replace NaN with empty string in choice columns\n    choice_cols = ['choice_A', 'choice_B', 'choice_C', 'choice_D', 'choice_E']\n    for col in choice_cols:\n        df[col] = df[col].fillna('')\n\n    # Convert to list of MultipleChoiceData objects\n    data = []\n    for _, row in df.iterrows():\n        choices = [row['choice_A'], row['choice_B'], row['choice_C'],\n                  row['choice_D'], row['choice_E']]\n        data.append(MultipleChoiceData(\n            task_id=row['task_id'],\n            question=row['question'],\n            choices=choices,\n            answer=row['answer']\n        ))\n    return data\n\n# Custom Dataset class for multiple choice\nclass MultipleChoiceDataset(torch.utils.data.Dataset):\n    def __init__(self, data: List[MultipleChoiceData], tokenizer, max_length: int = 512):\n        self.data = data\n        self.tokenizer = tokenizer\n        self.max_length = max_length\n        self.label_map = {'A': 0, 'B': 1, 'C': 2, 'D': 3, 'E': 4}\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        item = self.data[idx]\n\n        # Prepare inputs: repeat question for each choice\n        questions = [item.question] * 5\n        choices = item.choices\n\n        # Tokenize\n        encoding = self.tokenizer(\n            questions,\n            choices,\n            max_length=self.max_length,\n            padding='max_length',\n            truncation=True,\n            return_tensors='pt'\n        )\n\n        # Remove batch dimension from tokenizer output\n        encoding = {k: v.squeeze(0) for k, v in encoding.items()}\n        encoding['labels'] = torch.tensor(self.label_map[item.answer])\n\n        return encoding\n\n# Compute metrics function (Not used in CustomLoraTrainer directly, but good practice)\ndef compute_metrics(eval_pred):\n    predictions, labels = eval_pred\n    preds = np.argmax(predictions, axis=1)\n    accuracy = (preds == labels).mean()\n    return {\"accuracy\": accuracy}\n\n# Custom Multiple Choice model for Qwen2\nclass Qwen2ForMultipleChoice(nn.Module):\n    def __init__(self, model_name):\n        super().__init__()\n        self.model = AutoModel.from_pretrained(model_name)\n        self.dropout = nn.Dropout(0.1)\n        # Ensure classifier's input dimension matches the base model's hidden size\n        self.classifier = nn.Linear(self.model.config.hidden_size, 1)\n\n    def forward(\n        self,\n        input_ids=None,\n        attention_mask=None,\n        token_type_ids=None, # Usually not needed for models like Qwen2\n        position_ids=None, # Usually not needed unless specifically required\n        labels=None,\n        **kwargs # Catch any unexpected arguments\n    ):\n        num_choices = input_ids.shape[1] if input_ids is not None else kwargs.get(\"inputs_embeds\").shape[1]\n\n        # Reshape input for the base model\n        flat_input_ids = input_ids.view(-1, input_ids.size(-1)) if input_ids is not None else None\n        flat_attention_mask = attention_mask.view(-1, attention_mask.size(-1)) if attention_mask is not None else None\n        # Qwen2 typically doesn't use token_type_ids, so we can omit passing them\n        # flat_token_type_ids = token_type_ids.view(-1, token_type_ids.size(-1)) if token_type_ids is not None else None\n\n        # Pass only relevant inputs to the base model\n        model_inputs = {\n            \"input_ids\": flat_input_ids,\n            \"attention_mask\": flat_attention_mask,\n        }\n        # Conditionally add position_ids if provided and needed by the specific base model variant\n        # if position_ids is not None:\n        #     flat_position_ids = position_ids.view(-1, position_ids.size(-1))\n        #     model_inputs[\"position_ids\"] = flat_position_ids\n\n        outputs = self.model(**model_inputs)\n\n        # Use the hidden state of the first token (often [CLS] or equivalent)\n        # For Qwen2, pooling the last hidden state might be better, let's stick to first token for consistency with original code\n        pooled_output = outputs.last_hidden_state[:, 0] # Using last_hidden_state\n        # Alternative: If using outputs[0], it's usually last_hidden_state\n        # pooled_output = outputs[0][:, 0]\n\n        pooled_output = self.dropout(pooled_output)\n        logits = self.classifier(pooled_output)\n        reshaped_logits = logits.view(-1, num_choices)\n\n        loss = None\n        if labels is not None:\n            loss_fct = nn.CrossEntropyLoss()\n            loss = loss_fct(reshaped_logits, labels)\n\n        # Ensure we return a MultipleChoiceModelOutput object\n        return MultipleChoiceModelOutput(\n            loss=loss,\n            logits=reshaped_logits,\n            hidden_states=outputs.hidden_states, # Pass through from base model\n            attentions=outputs.attentions,     # Pass through from base model\n        )\n\n\n# Custom trainer to handle LoRA without using HF Trainer\nclass CustomLoraTrainer:\n    def __init__(self, model, train_dataset, eval_dataset, tokenizer,\n                 batch_size=4, num_epochs=3, learning_rate=5e-5,\n                 device=\"cpu\", # Primary device (e.g., \"cuda:0\")\n                 is_parallel=False): # Flag to know if model is wrapped\n        self.model = model # Model might be wrapped in DataParallel\n        self.train_dataset = train_dataset\n        self.eval_dataset = eval_dataset\n        self.tokenizer = tokenizer\n        self.batch_size = batch_size # This will be per device if using DataParallel\n        self.num_epochs = num_epochs\n        self.learning_rate = learning_rate\n        self.device = device # The primary device where data should be moved\n        self.is_parallel = is_parallel # Store whether the model is parallelized\n\n        # Move model to the primary device (DataParallel handles distribution)\n        self.model.to(self.device)\n\n        # Set up optimizer - parameters() works correctly with DataParallel\n        self.optimizer = optim.AdamW(self.model.parameters(), lr=self.learning_rate)\n\n        # Create dataloaders\n        self.train_dataloader = DataLoader(\n            train_dataset,\n            batch_size=self.batch_size, # Batch size per GPU\n            shuffle=True,\n            num_workers=4 # Increase num_workers for potentially faster loading\n        )\n\n        self.eval_dataloader = DataLoader(\n            eval_dataset,\n            batch_size=self.batch_size, # Batch size per GPU\n            num_workers=4\n        )\n\n    def train(self):\n        self.model.train()\n        total_loss = 0\n\n        progress_bar = tqdm(self.train_dataloader, desc=\"Training\")\n        for batch in progress_bar:\n            # Move batch to the *primary* device. DataParallel handles scattering.\n            batch = {k: v.to(self.device) for k, v in batch.items()}\n\n            # Zero gradients\n            self.optimizer.zero_grad()\n\n            # Forward pass\n            outputs = self.model(\n                input_ids=batch['input_ids'],\n                attention_mask=batch['attention_mask'],\n                labels=batch['labels']\n            )\n\n            loss = outputs.loss\n            # If using DataParallel, loss might be a tensor containing loss per device. Average it.\n            if self.is_parallel and loss.ndim > 0: # Check if loss is not scalar\n                 loss = loss.mean()\n\n            total_loss += loss.item()\n\n            # Backward pass\n            loss.backward()\n            self.optimizer.step()\n\n            # Update progress bar\n            progress_bar.set_postfix({\"loss\": loss.item()})\n\n        return total_loss / len(self.train_dataloader)\n\n    def evaluate(self):\n        self.model.eval()\n        total_loss = 0\n        all_preds = []\n        all_labels = []\n\n        with torch.no_grad():\n            progress_bar = tqdm(self.eval_dataloader, desc=\"Evaluating\")\n            for batch in progress_bar:\n                # Move batch to the primary device\n                batch = {k: v.to(self.device) for k, v in batch.items()}\n\n                # Forward pass\n                outputs = self.model(\n                    input_ids=batch['input_ids'],\n                    attention_mask=batch['attention_mask'],\n                    labels=batch['labels']\n                )\n\n                loss = outputs.loss\n                 # Average loss if needed (similar to training loop)\n                if self.is_parallel and loss.ndim > 0:\n                     loss = loss.mean()\n\n                total_loss += loss.item()\n\n                # Get predictions (logits are gathered on the primary device by DataParallel)\n                preds = torch.argmax(outputs.logits, dim=1)\n                all_preds.extend(preds.cpu().numpy())\n                all_labels.extend(batch['labels'].cpu().numpy())\n\n                # Update progress bar\n                progress_bar.set_postfix({\"eval_loss\": loss.item()})\n\n        # Calculate accuracy\n        accuracy = (np.array(all_preds) == np.array(all_labels)).mean()\n\n        return {\n            \"loss\": total_loss / len(self.eval_dataloader),\n            \"accuracy\": accuracy\n        }\n\n    def train_and_evaluate(self):\n        best_accuracy = 0\n        for epoch in range(self.num_epochs):\n            print(f\"\\nEpoch {epoch+1}/{self.num_epochs}\")\n\n            # Train\n            train_loss = self.train()\n            print(f\"Train loss: {train_loss:.4f}\")\n\n            # Evaluate\n            eval_results = self.evaluate()\n            print(f\"Eval loss: {eval_results['loss']:.4f}, Accuracy: {eval_results['accuracy']:.4f}\")\n\n            # Track best accuracy (saving happens outside in the main loop)\n            if eval_results['accuracy'] > best_accuracy:\n                best_accuracy = eval_results['accuracy']\n                print(f\"New best accuracy during this fold's training: {best_accuracy:.4f}\")\n\n        # Return the best accuracy achieved during this fold's training\n        return best_accuracy","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-03-30T16:45:41.403281Z","iopub.execute_input":"2025-03-30T16:45:41.403486Z","iopub.status.idle":"2025-03-30T16:45:41.416507Z","shell.execute_reply.started":"2025-03-30T16:45:41.403469Z","shell.execute_reply":"2025-03-30T16:45:41.415893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile trainer.py\nimport torch\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader\nimport numpy as np\nfrom tqdm.auto import tqdm\nfrom transformers import get_linear_schedule_with_warmup, get_cosine_schedule_with_warmup\n\nclass CustomLoraTrainer:\n    def __init__(self, model, train_dataset, eval_dataset, tokenizer,\n                 batch_size=4, num_epochs=3, learning_rate=5e-5,\n                 weight_decay=0.01, lr_scheduler_type=None, num_warmup_steps=0,\n                 device=\"cpu\", num_gpus=0, data_collator=None):\n        self.model = model\n        self.train_dataset = train_dataset\n        self.eval_dataset = eval_dataset\n        self.tokenizer = tokenizer\n        self.batch_size = batch_size\n        self.num_epochs = num_epochs\n        self.learning_rate = learning_rate\n        self.weight_decay = weight_decay\n        self.lr_scheduler_type = lr_scheduler_type\n        self.num_warmup_steps = num_warmup_steps\n        self.device = device\n        self.num_gpus = num_gpus\n        self.data_collator = data_collator\n\n        # Move model to device - DataParallel handles distribution\n        self.model.to(self.device)\n\n        # Set up optimizer with weight decay\n        self.optimizer = optim.AdamW(\n            self.model.parameters(),\n            lr=self.learning_rate,\n            weight_decay=self.weight_decay\n        )\n\n        # Create dataloaders using data_collator if provided\n        self.train_dataloader = DataLoader(\n            train_dataset,\n            batch_size=self.batch_size,\n            shuffle=True,\n            num_workers=4,\n            pin_memory=True,\n            collate_fn=self.data_collator\n        )\n\n        self.eval_dataloader = DataLoader(\n            eval_dataset,\n            batch_size=self.batch_size,\n            shuffle=False,\n            num_workers=4,\n            pin_memory=True,\n            collate_fn=self.data_collator\n        )\n        \n        # Set up scheduler if requested\n        self.scheduler = None\n        if self.lr_scheduler_type:\n            num_training_steps = len(self.train_dataloader) * self.num_epochs\n            warmup_steps = self.num_warmup_steps if self.num_warmup_steps > 0 else int(0.1 * num_training_steps)\n            \n            print(f\"Setting up {self.lr_scheduler_type} scheduler with {warmup_steps} warmup steps out of {num_training_steps} total steps\")\n            \n            if self.lr_scheduler_type.lower() == \"linear\":\n                self.scheduler = get_linear_schedule_with_warmup(\n                    self.optimizer,\n                    num_warmup_steps=warmup_steps,\n                    num_training_steps=num_training_steps\n                )\n            elif self.lr_scheduler_type.lower() == \"cosine\":\n                self.scheduler = get_cosine_schedule_with_warmup(\n                    self.optimizer,\n                    num_warmup_steps=warmup_steps,\n                    num_training_steps=num_training_steps\n                )\n\n    def train(self):\n        self.model.train()\n        total_loss = 0\n\n        progress_bar = tqdm(self.train_dataloader, desc=\"Training\")\n        for batch in progress_bar:\n            # Handle special keys separately from the main tensors\n            inputs = {}\n            special_keys = ['cls_positions', 'valid_choice_mask', 'example_ids']\n            \n            # Move regular tensors to device\n            for k, v in batch.items():\n                if k not in special_keys:\n                    inputs[k] = v.to(self.device) if isinstance(v, torch.Tensor) else v\n            \n            # Handle special keys\n            for key in special_keys:\n                if key in batch:\n                    if key == 'example_ids':\n                        inputs[key] = batch[key]\n                    else:\n                        inputs[key] = batch[key].to(self.device)\n\n            # Zero gradients\n            self.optimizer.zero_grad()\n\n            # Forward pass\n            outputs = self.model(**inputs)\n            loss = outputs.loss\n            \n            # Handle loss reduction for DataParallel - same as DeBERTa\n            if hasattr(loss, 'dim') and loss.dim() > 0:\n                loss = loss.mean()\n\n            total_loss += loss.item()\n\n            # Backward pass\n            loss.backward()\n            self.optimizer.step()\n            \n            # Step scheduler if available\n            if self.scheduler:\n                self.scheduler.step()\n\n            # Update progress bar\n            progress_bar.set_postfix({\"loss\": loss.item()})\n\n        return total_loss / len(self.train_dataloader)\n\n    def evaluate(self):\n        self.model.eval()\n        total_loss = 0\n        all_preds = []\n        all_labels = []\n\n        with torch.no_grad():\n            progress_bar = tqdm(self.eval_dataloader, desc=\"Evaluating\")\n            for batch in progress_bar:\n                # Handle special keys separately from the main tensors\n                inputs = {}\n                special_keys = ['cls_positions', 'valid_choice_mask', 'example_ids']\n                \n                # Move regular tensors to device\n                for k, v in batch.items():\n                    if k not in special_keys:\n                        inputs[k] = v.to(self.device) if isinstance(v, torch.Tensor) else v\n                \n                # Handle special keys\n                for key in special_keys:\n                    if key in batch:\n                        if key == 'example_ids':\n                            inputs[key] = batch[key]\n                        else:\n                            inputs[key] = batch[key].to(self.device)\n\n                # Forward pass\n                outputs = self.model(**inputs)\n                loss = outputs.loss\n                \n                # Handle loss reduction for DataParallel\n                if hasattr(loss, 'dim') and loss.dim() > 0:\n                    loss = loss.mean()\n\n                total_loss += loss.item()\n\n                # Get predictions\n                preds = torch.argmax(outputs.logits, dim=1)\n                all_preds.extend(preds.cpu().numpy())\n                all_labels.extend(inputs['labels'].cpu().numpy())\n\n                # Update progress bar\n                progress_bar.set_postfix({\"eval_loss\": loss.item()})\n\n        # Calculate accuracy\n        accuracy = (np.array(all_preds) == np.array(all_labels)).mean()\n\n        return {\n            \"loss\": total_loss / len(self.eval_dataloader),\n            \"accuracy\": accuracy\n        }\n\n    def train_and_evaluate(self):\n        best_accuracy = 0\n        for epoch in range(self.num_epochs):\n            print(f\"\\nEpoch {epoch+1}/{self.num_epochs}\")\n\n            # Train\n            train_loss = self.train()\n            print(f\"Train loss: {train_loss:.4f}\")\n\n            # Evaluate\n            eval_results = self.evaluate()\n            print(f\"Eval loss: {eval_results['loss']:.4f}, Accuracy: {eval_results['accuracy']:.4f}\")\n\n            # Track best accuracy\n            if eval_results['accuracy'] > best_accuracy:\n                best_accuracy = eval_results['accuracy']\n                print(f\"New best accuracy during this fold's training: {best_accuracy:.4f}\")\n\n        # Return the best accuracy achieved during this fold's training\n        return best_accuracy","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-03-30T16:45:41.58149Z","iopub.execute_input":"2025-03-30T16:45:41.58173Z","iopub.status.idle":"2025-03-30T16:45:41.586319Z","shell.execute_reply.started":"2025-03-30T16:45:41.581712Z","shell.execute_reply":"2025-03-30T16:45:41.585702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%writefile main.py\nimport torch\nimport numpy as np\nimport os\nfrom sklearn.model_selection import KFold\nfrom transformers import AutoTokenizer\nimport torch.nn as nn\n\nfrom config import Config\nfrom data import load_and_preprocess_data, MultipleChoiceDataset, CustomDataCollator\nfrom model import Qwen2ForMultipleChoice\nfrom trainer import CustomLoraTrainer\nfrom lora_utils import apply_lora_to_model, save_lora_weights\n\ndef main():\n    # Load configuration\n    config = Config()\n    \n    # Load data\n    if not os.path.exists(config.data_path):\n        raise FileNotFoundError(f\"Data file not found at {config.data_path}\")\n    data = load_and_preprocess_data(config.data_path)\n\n    # Initialize tokenizer\n    tokenizer = AutoTokenizer.from_pretrained(config.model_name)\n    # Qwen2 tokenizer might not have a pad token by default\n    if tokenizer.pad_token is None:\n        tokenizer.pad_token = tokenizer.eos_token\n        print(\"Set tokenizer pad_token to eos_token\")\n\n    # Prepare K-fold cross-validation\n    kf = KFold(n_splits=config.num_folds, shuffle=True, random_state=42)\n    fold_results = []\n\n    # Convert data to array for indexing\n    data_array = np.array(data)\n\n    # Cross-validation loop\n    for fold, (train_idx, val_idx) in enumerate(kf.split(data_array)):\n        print(f\"\\n===== Training Fold {fold + 1}/{config.num_folds} =====\")\n\n        # Split data\n        train_data = data_array[train_idx].tolist()\n        val_data = data_array[val_idx].tolist()\n\n        # Create datasets with config to enable special tokens\n        train_dataset = MultipleChoiceDataset(train_data, tokenizer, max_length=config.max_length, config=config)\n        val_dataset = MultipleChoiceDataset(val_data, tokenizer, max_length=config.max_length, config=config)\n\n        # Load model - pass num_choices from config\n        model = Qwen2ForMultipleChoice(config.model_name, num_choices=config.num_choices)\n\n        print(\"Unfreeze base model parameters...\")\n        for param in model.model.parameters():\n            param.requires_grad = True\n\n        if config.use_lora:\n            # Apply LoRA to the model\n            print(\"Applying LoRA to model...\")\n            model = apply_lora_to_model(model, config)\n        else:\n            # Without LoRA, just keep the head trainable as before\n            print(\"Using standard fine-tuning with frozen backbone...\")\n            for param in model.head.parameters():\n                param.requires_grad = True\n            \n            # Print parameter status\n            trainable_params = sum(p.numel() for p in model.parameters() if p.requires_grad)\n            total_params = sum(p.numel() for p in model.parameters())\n            print(f\"Trainable parameters: {trainable_params:,d} ({100 * trainable_params / total_params:.2f}%)\")\n            print(f\"Total parameters: {total_params:,d}\")\n\n        # Apply DataParallel if using multiple GPUs\n        if config.num_gpus > 1:\n            print(f\"Wrapping model with nn.DataParallel across {config.num_gpus} GPUs\")\n            model = nn.DataParallel(model)\n\n        # Move model to device\n        model.to(config.device)\n\n        # Use custom data collator\n        data_collator = CustomDataCollator(tokenizer=tokenizer)\n\n        # Use our custom trainer with scheduler and weight decay\n        trainer = CustomLoraTrainer(\n            model=model,\n            train_dataset=train_dataset,\n            eval_dataset=val_dataset,\n            tokenizer=tokenizer,\n            data_collator=data_collator,\n            batch_size=config.batch_size,\n            num_epochs=config.num_epochs,\n            learning_rate=config.learning_rate,\n            weight_decay=config.weight_decay,\n            lr_scheduler_type=config.lr_scheduler_type,\n            num_warmup_steps=config.num_warmup_steps,\n            device=config.device,\n            num_gpus=config.num_gpus\n        )\n\n        # Train and evaluate\n        fold_best_accuracy = trainer.train_and_evaluate()\n        fold_results.append(fold_best_accuracy)\n\n        # Save model for this fold\n        model_save_path = f\"{config.output_dir}/fold_{fold}/best_model\"\n        os.makedirs(model_save_path, exist_ok=True)\n        print(f\"Saving model for fold {fold} to {model_save_path}\")\n\n        if config.use_lora:\n            # Save LoRA weights\n            print(f\"Saving LoRA weights to {model_save_path}\")\n            if isinstance(model, nn.DataParallel):\n                save_lora_weights(model.module, model_save_path)\n            else:\n                save_lora_weights(model, model_save_path)\n        else:\n            # Save the model state dict\n            model_state_dict = model.module.state_dict() if isinstance(model, nn.DataParallel) else model.state_dict()\n            torch.save(model_state_dict, f\"{model_save_path}/model.pth\")\n\n        # Save tokenizer\n        tokenizer.save_pretrained(model_save_path)\n\n        # Clean up GPU memory\n        del model, trainer, train_dataset, val_dataset\n        if torch.cuda.is_available():\n            torch.cuda.empty_cache()\n\n    # Print overall results\n    mean_accuracy = np.mean(fold_results)\n    std_accuracy = np.std(fold_results)\n    print(\"\\n===== Cross-validation Results =====\")\n    print(f\"Fold accuracies: {fold_results}\")\n    print(f\"Mean Accuracy across folds: {mean_accuracy:.4f}\")\n    print(f\"Standard Deviation across folds: {std_accuracy:.4f}\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2025-03-30T16:45:41.591919Z","iopub.execute_input":"2025-03-30T16:45:41.592124Z","iopub.status.idle":"2025-03-30T16:45:41.602018Z","shell.execute_reply.started":"2025-03-30T16:45:41.592106Z","shell.execute_reply":"2025-03-30T16:45:41.601443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%run main.py","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-30T16:45:41.602787Z","iopub.execute_input":"2025-03-30T16:45:41.602987Z","execution_failed":"2025-03-30T18:34:58.204Z"}},"outputs":[],"execution_count":null}]}