{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":39763,"databundleVersionId":11756775,"sourceType":"competition"},{"sourceId":232758274,"sourceType":"kernelVersion"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Geophysical Waveform Inversion - EDA Notebook**","metadata":{}},{"cell_type":"markdown","source":"# Setup and Configuration","metadata":{}},{"cell_type":"code","source":"import os\nimport sys\nimport json\nimport glob\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport torch\nimport pandas as pd\nfrom torchvision.transforms import Compose\nimport torch.nn as nn\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nfrom typing import List\nimport logging\nimport csv\nimport torch.nn.functional as F\nimport csv\n\n# Configure logging\nlogging.basicConfig(format='[%(levelname)s] %(message)s', level=logging.INFO)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T09:33:31.993702Z","iopub.execute_input":"2025-04-12T09:33:31.993950Z","iopub.status.idle":"2025-04-12T09:33:40.550071Z","shell.execute_reply.started":"2025-04-12T09:33:31.993928Z","shell.execute_reply":"2025-04-12T09:33:40.549110Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_dataset_config(config_path, dataset_name):\n    \"\"\"Loads normalization parameters from dataset_config.json.\"\"\"\n    try:\n        with open(config_path) as f:\n            ctx = json.load(f)[dataset_name]\n        print(f\"Loaded config for dataset: {dataset_name}\")\n        return ctx\n    except FileNotFoundError:\n        print(f\"Error: {config_path} not found.\")\n        sys.exit(1)\n    except KeyError:\n        print(f\"Error: Dataset '{dataset_name}' not found in {config_path}.\")\n        sys.exit(1)\n\ndef get_transforms(ctx, k):\n    \"\"\"Gets the transformations for data and label based on test.py.\"\"\"\n    log_data_min = T.log_transform(ctx['data_min'], k=k)\n    log_data_max = T.log_transform(ctx['data_max'], k=k)\n    transform_data = Compose([\n        T.LogTransform(k=k),\n        T.MinMaxNormalize(log_data_min, log_data_max),\n    ])\n\n    return transform_data\n    \n# ================================================================\n# Flexible Exploration for Any Family\n# Auto-Skip empty folders\n# ================================================================\n\ndef explore_family(folder_name):\n    folder_path = os.path.join(TRAIN_DIR, folder_name)\n    print(f\"\\nExploring {folder_name} Dataset\")\n    print(\"Available Files:\", os.listdir(folder_path))\n\n    seis_files = sorted([f for f in os.listdir(folder_path) if f.startswith('seis')])\n    vel_files = sorted([f for f in os.listdir(folder_path) if f.startswith('vel')])\n\n    print(f\"Found {len(seis_files)} Seismic files\")\n    print(f\"Found {len(vel_files)} Velocity files\")\n\n    # Check before loading\n    if seis_files and vel_files:\n        example_seis = load_npy(os.path.join(folder_path, seis_files[0]))\n        example_vel = load_npy(os.path.join(folder_path, vel_files[0]))\n        example_vel = np.squeeze(example_vel)\n\n        print(\"Seismic Shape:\", example_seis.shape)\n        print(\"Velocity Shape:\", example_vel.shape)\n    else:\n        print(\"Skipping... No seismic or velocity files found.\")\n\n# ================================================================\n# Helper to Load Numpy file\n# ================================================================\ndef load_npy(file_path):\n    return np.load(file_path)\n    \n# =============================================================================\n# 1. Data Preparation\n# =============================================================================\ndef collect_input_files(data_dir: str) -> list:\n    \"\"\"\n    Recursively search for .npy files in data_dir that contain 'seis' or 'data' in their filename.\n    \"\"\"\n    return [f for f in Path(data_dir).rglob(\"*.npy\") if (\"seis\" in f.stem) or (\"data\" in f.stem)]\n\ndef map_input_to_output(input_files: list) -> list:\n    \"\"\"\n    Map each input file to its corresponding output file by replacing keywords.\n    \"\"\"\n    return [Path(str(f).replace(\"seis\", \"vel\").replace(\"data\", \"model\")) for f in input_files]\n\n# Define training sample directory\nTRAIN_DIR = \"/kaggle/input/waveform-inversion/train_samples\"\ninputs_all = collect_input_files(TRAIN_DIR)\noutputs_all = map_input_to_output(inputs_all)\n\n# Check all output files exist\nassert all(f.exists() for f in outputs_all)\n\n# Split dataset into training and validation based on sampling frequency\ntrain_inputs = [inputs_all[i] for i in range(0, len(inputs_all), 2)]\nvalid_inputs = [f for f in inputs_all if f not in train_inputs]\ntrain_outputs = map_input_to_output(train_inputs)\nvalid_outputs = map_input_to_output(valid_inputs)\n\n# =============================================================================\n# 2. Dataset Definition\n# =============================================================================\nclass SeismicDataset(Dataset):\n    \"\"\"\n    Dataset handling seismic files with multiple examples per file.\n    \"\"\"\n    def __init__(self, in_files: list, out_files: list, examples_per_file: int = 500):\n        assert len(in_files) == len(out_files)\n        self.in_files = in_files\n        self.out_files = out_files\n        self.examples_per_file = examples_per_file\n\n    def __len__(self):\n        return len(self.in_files) * self.examples_per_file\n\n    def __getitem__(self, idx: int):\n        file_index = idx // self.examples_per_file\n        sample_index = idx % self.examples_per_file\n\n        # Memory map the file to reduce memory usage\n        x_data = np.load(self.in_files[file_index], mmap_mode=\"r\")\n        y_data = np.load(self.out_files[file_index], mmap_mode=\"r\")\n        try:\n            return x_data[sample_index].copy(), y_data[sample_index].copy()\n        finally:\n            del x_data, y_data\n\n# Create DataLoaders for training and validation\ntrain_dataset = SeismicDataset(train_inputs, train_outputs, examples_per_file=500)\nvalid_dataset = SeismicDataset(valid_inputs, valid_outputs, examples_per_file=500)\n\ntrain_loader = DataLoader(\n    train_dataset, batch_size=64, shuffle=True, pin_memory=True,\n    drop_last=True, num_workers=4, persistent_workers=True\n)\nvalid_loader = DataLoader(\n    valid_dataset, batch_size=64, shuffle=False, pin_memory=True,\n    drop_last=False, num_workers=4, persistent_workers=True\n)\n\n# =============================================================================\n# 3. Model Architecture: SmartConvNet\n# =============================================================================\nclass SmartConvNet(nn.Module):\n    \"\"\"A convolutional network with adaptive pooling and dense layers.\"\"\"\n    def __init__(self, input_channels: int = 5, output_size: int = 70 * 70):\n        super().__init__()\n        # Convolutional feature extractor\n        self.feature_extractor = nn.Sequential(\n            nn.Conv2d(input_channels, 16, kernel_size=3, stride=2, padding=1),  # spatial reduction\n            nn.BatchNorm2d(16),\n            nn.ReLU(),\n            nn.Conv2d(16, 32, kernel_size=3, stride=2, padding=1),  # further reduction\n            nn.BatchNorm2d(32),\n            nn.ReLU(),\n            nn.Conv2d(32, 64, kernel_size=3, stride=1, padding=1),\n            nn.BatchNorm2d(64),\n            nn.ReLU(),\n        )\n        # Pool output to a fixed size\n        self.pool = nn.AdaptiveAvgPool2d((7, 7))\n        # Fully connected head\n        self.fc = nn.Sequential(\n            nn.Linear(64 * 7 * 7, 512),\n            nn.ReLU(),\n            nn.Dropout(0.3),\n            nn.Linear(512, output_size),\n        )\n\n    def forward(self, x):\n        batch_size = x.shape[0]\n        feat = self.feature_extractor(x)\n        pooled = self.pool(feat)\n        flat = pooled.view(batch_size, -1)\n        out = self.fc(flat)\n        # Reshape output to (batch_size, 1, 70, 70) and apply scaling and bias\n        return out.view(batch_size, 1, 70, 70) * 1000 + 1500\n\n# -----------------------------------------------------------------------------\n# Residual Block with Squeeze-and-Excitation (SE) Module\n# -----------------------------------------------------------------------------\nclass ResidualBlock(nn.Module):\n    \"\"\"\n    A residual block that optionally uses a squeeze-and-excitation (SE) module to\n    adaptively weight the channels.\n    \"\"\"\n    def __init__(self, in_channels, out_channels, stride=1, use_se=True):\n        super().__init__()\n        self.use_se = use_se\n        \n        self.conv1 = nn.Conv2d(in_channels, out_channels, kernel_size=3, \n                               stride=stride, padding=1, bias=False)\n        self.bn1 = nn.BatchNorm2d(out_channels)\n        self.relu = nn.ReLU(inplace=True)\n        \n        self.conv2 = nn.Conv2d(out_channels, out_channels, kernel_size=3, \n                               stride=1, padding=1, bias=False)\n        self.bn2 = nn.BatchNorm2d(out_channels)\n        \n        # If there's a change in dimensions, adjust the shortcut (residual path)\n        self.downsample = None\n        if stride != 1 or in_channels != out_channels:\n            self.downsample = nn.Sequential(\n                nn.Conv2d(in_channels, out_channels, kernel_size=1,\n                          stride=stride, bias=False),\n                nn.BatchNorm2d(out_channels)\n            )\n        \n        # Squeeze-and-Excitation module for channel attention\n        if self.use_se:\n            self.se = nn.Sequential(\n                nn.AdaptiveAvgPool2d(1),\n                nn.Conv2d(out_channels, out_channels // 16, kernel_size=1),\n                nn.ReLU(inplace=True),\n                nn.Conv2d(out_channels // 16, out_channels, kernel_size=1),\n                nn.Sigmoid()\n            )\n    \n    def forward(self, x):\n        identity = x\n        \n        out = self.relu(self.bn1(self.conv1(x)))\n        out = self.bn2(self.conv2(out))\n        \n        if self.use_se:\n            se_weight = self.se(out)\n            out = out * se_weight\n        \n        if self.downsample is not None:\n            identity = self.downsample(x)\n            \n        out += identity\n        return self.relu(out)\n\n# -----------------------------------------------------------------------------\n# ComplexConvNet: Enhanced and Deeper Model Architecture\n# -----------------------------------------------------------------------------\nclass ComplexConvNet(nn.Module):\n    \"\"\"\n    A more complex deep convolutional network that uses a stem followed by a series of\n    residual blocks with SE modules. The network employs global pooling and dense layers\n    to produce an output of shape (batch_size, 1, 70, 70) with the desired scaling.\n    \"\"\"\n    def __init__(self, input_channels=5, output_size=70*70):\n        super().__init__()\n        # Stem: initial feature extractor\n        self.stem = nn.Sequential(\n            nn.Conv2d(input_channels, 32, kernel_size=3, stride=2, padding=1, bias=False),\n            nn.BatchNorm2d(32),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(32, 32, kernel_size=3, stride=1, padding=1, bias=False),\n            nn.BatchNorm2d(32),\n            nn.ReLU(inplace=True)\n        )\n        \n        # Residual layers with increasing feature channels, each using SE\n        self.layer1 = nn.Sequential(\n            ResidualBlock(32, 64, stride=2, use_se=True),\n            ResidualBlock(64, 64, stride=1, use_se=True)\n        )\n        self.layer2 = nn.Sequential(\n            ResidualBlock(64, 128, stride=2, use_se=True),\n            ResidualBlock(128, 128, stride=1, use_se=True)\n        )\n        self.layer3 = nn.Sequential(\n            ResidualBlock(128, 256, stride=2, use_se=True),\n            ResidualBlock(256, 256, stride=1, use_se=True)\n        )\n        self.layer4 = nn.Sequential(\n            ResidualBlock(256, 512, stride=2, use_se=True),\n            ResidualBlock(512, 512, stride=1, use_se=True)\n        )\n        \n        # Global Pooling to get fixed spatial dimensions regardless of input size\n        self.global_pool = nn.AdaptiveAvgPool2d((4, 4))\n        \n        # Dense (fully connected) layers for final processing\n        self.fc = nn.Sequential(\n            nn.Flatten(),\n            nn.Linear(512 * 4 * 4, 2048),\n            nn.GELU(),\n            nn.Dropout(0.5),\n            nn.Linear(2048, 1024),\n            nn.GELU(),\n            nn.Dropout(0.3),\n            nn.Linear(1024, output_size)\n        )\n        \n    def forward(self, x):\n        batch_size = x.shape[0]\n        x = self.stem(x)\n        x = self.layer1(x)\n        x = self.layer2(x)\n        x = self.layer3(x)\n        x = self.layer4(x)\n        x = self.global_pool(x)\n        x = self.fc(x)\n        \n        # Reshape output to (batch_size, 1, 70, 70) and apply scaling and offset\n        return x.view(batch_size, 1, 70, 70) * 1000 + 1500\n\nclass TestDataset(Dataset):\n    \"\"\"\n    Dataset for test files that returns the test data and its identifier.\n    \"\"\"\n    def __init__(self, files: list):\n        self.files = files\n\n    def __len__(self):\n        return len(self.files)\n\n    def __getitem__(self, idx: int):\n        file_path = self.files[idx]\n        return np.load(file_path), file_path.stem\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:22:15.909016Z","iopub.execute_input":"2025-04-12T10:22:15.909389Z","iopub.status.idle":"2025-04-12T10:22:16.093249Z","shell.execute_reply.started":"2025-04-12T10:22:15.909356Z","shell.execute_reply":"2025-04-12T10:22:16.092255Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Root Paths\nBASE_DIR = '/kaggle/input/waveform-inversion'\nTRAIN_DIR = os.path.join(BASE_DIR, 'train_samples')\nTEST_DIR = os.path.join(BASE_DIR, 'test')\n\nprint(\"Train Folders:\", os.listdir(TRAIN_DIR))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T09:35:48.131865Z","iopub.execute_input":"2025-04-12T09:35:48.132135Z","iopub.status.idle":"2025-04-12T09:35:48.148553Z","shell.execute_reply.started":"2025-04-12T09:35:48.132113Z","shell.execute_reply":"2025-04-12T09:35:48.147707Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Explore CurveFault_A as Example\n","metadata":{}},{"cell_type":"code","source":"curve_fault_a_path = os.path.join(TRAIN_DIR, 'CurveFault_A')\nprint(\"Files in CurveFault_A:\", os.listdir(curve_fault_a_path))\nseis_file = os.path.join(curve_fault_a_path, 'seis2_1_0.npy')\nvel_file = os.path.join(curve_fault_a_path, 'vel2_1_0.npy')\n\nseis = load_npy(seis_file)\nvel = load_npy(vel_file)\n\nprint(\"Seismic Data shape:\", seis.shape)  \nprint(\"Velocity Data shape:\", vel.shape)\n\nvel = np.squeeze(vel)  \n\nprint(\"Velocity Shape after squeeze:\", vel.shape)\n\nsample_id = 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T09:35:48.512014Z","iopub.execute_input":"2025-04-12T09:35:48.512311Z","iopub.status.idle":"2025-04-12T09:35:52.546292Z","shell.execute_reply.started":"2025-04-12T09:35:48.512267Z","shell.execute_reply":"2025-04-12T09:35:52.545543Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Velocity Map","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.title(f\"Velocity Map (Ground Truth) - Sample {sample_id}\")\nsns.heatmap(vel[sample_id], cmap='viridis')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T09:35:52.547256Z","iopub.execute_input":"2025-04-12T09:35:52.547537Z","iopub.status.idle":"2025-04-12T09:35:53.240820Z","shell.execute_reply.started":"2025-04-12T09:35:52.547516Z","shell.execute_reply":"2025-04-12T09:35:53.240019Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Seismic Data","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.title(f\"Seismic Data - Batch 0, Source 0\")\nplt.imshow(seis[0, 0], aspect='auto', cmap='seismic')\nplt.colorbar(label=\"Amplitude\")\nplt.xlabel(\"Receivers\")\nplt.ylabel(\"Timesteps\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T09:35:53.242190Z","iopub.execute_input":"2025-04-12T09:35:53.242449Z","iopub.status.idle":"2025-04-12T09:35:53.562113Z","shell.execute_reply.started":"2025-04-12T09:35:53.242427Z","shell.execute_reply":"2025-04-12T09:35:53.561303Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Explore All Families","metadata":{}},{"cell_type":"code","source":"families = os.listdir(TRAIN_DIR)\n\nfor fam in families:\n    explore_family(fam)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-12T09:35:53.563183Z","iopub.execute_input":"2025-04-12T09:35:53.563435Z","iopub.status.idle":"2025-04-12T09:36:03.468738Z","shell.execute_reply.started":"2025-04-12T09:35:53.563414Z","shell.execute_reply":"2025-04-12T09:36:03.468020Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Test Set Exploration=","metadata":{}},{"cell_type":"code","source":"test_files = sorted(os.listdir(TEST_DIR))\n\nprint(\"\\nSample Test File:\", test_files[0])\nsample_test = load_npy(os.path.join(TEST_DIR, test_files[0]))\nprint(\"Test Sample Shape:\", sample_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T09:36:03.469490Z","iopub.execute_input":"2025-04-12T09:36:03.469779Z","iopub.status.idle":"2025-04-12T09:36:03.934859Z","shell.execute_reply.started":"2025-04-12T09:36:03.469755Z","shell.execute_reply":"2025-04-12T09:36:03.934095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.title(\"Test Seismic Sample Visualization\")\n# Showing full 2D seismic data → shape (Receivers, Timesteps)\nplt.imshow(sample_test[0], aspect='auto', cmap='seismic')  \nplt.colorbar()\nplt.xlabel(\"Timesteps\")\nplt.ylabel(\"Receivers\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T09:36:03.935599Z","iopub.execute_input":"2025-04-12T09:36:03.935865Z","iopub.status.idle":"2025-04-12T09:36:04.225604Z","shell.execute_reply.started":"2025-04-12T09:36:03.935830Z","shell.execute_reply":"2025-04-12T09:36:04.224804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =============================================================================\n# 4. Device Setup and Model Initialization\n# =============================================================================\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = ComplexConvNet().to(device)\nprint(\"Using device:\", device)\n\n# =============================================================================\n# 5. Training Loop\n# =============================================================================\ncriterion = nn.L1Loss()\noptimizer = torch.optim.AdamW(model.parameters(), lr=1e-4, weight_decay=1e-3)\nn_epochs = 50\n\ntraining_history = []\n\nfor epoch in range(1, n_epochs + 1):\n    print(f\"[{epoch:02d}] Starting training\")\n    \n    # ----- Training Phase -----\n    model.train()\n    epoch_train_losses = []\n    for batch_inputs, batch_targets in tqdm(train_loader, desc=\"Training\", leave=False):\n        batch_inputs = batch_inputs.to(device)\n        batch_targets = batch_targets.to(device)\n        \n        optimizer.zero_grad()\n        predictions = model(batch_inputs)\n        loss = criterion(predictions, batch_targets)\n        loss.backward()\n        optimizer.step()\n        \n        epoch_train_losses.append(loss.item())\n    avg_train_loss = np.mean(epoch_train_losses)\n    print(\"Train loss: {:.5f}\".format(avg_train_loss))\n\n    # ----- Validation Phase -----\n    model.eval()\n    epoch_valid_losses = []\n    for batch_inputs, batch_targets in tqdm(valid_loader, desc=\"Validation\", leave=False):\n        batch_inputs = batch_inputs.to(device)\n        batch_targets = batch_targets.to(device)\n        \n        with torch.inference_mode():\n            predictions = model(batch_inputs)\n        loss = criterion(predictions, batch_targets)\n        epoch_valid_losses.append(loss.item())\n    avg_valid_loss = np.mean(epoch_valid_losses)\n    print(\"Valid loss: {:.5f}\".format(avg_valid_loss))\n    \n    training_history.append({\n        \"train\": avg_train_loss,\n        \"valid\": avg_valid_loss\n    })\n\n    # ----- Plot Example Outputs Every 4 Epochs -----\n    if epoch % 4 == 0:\n        sample_true = batch_targets[0, 0].detach().cpu()\n        sample_pred = predictions[0, 0].detach().cpu()\n        fig, axs = plt.subplots(1, 2, figsize=(5, 2.5))\n        fig.suptitle(f\"Epoch {epoch} | Valid: {avg_valid_loss:.5f}\")\n        axs[0].imshow(sample_true)\n        axs[0].set_title(\"Ground Truth\")\n        axs[1].imshow(sample_pred)\n        axs[1].set_title(\"Prediction\")\n        plt.show()\n\n# Plot training history\npd.DataFrame(training_history).plot(title=\"Training and Validation Loss History\");\n\n# =============================================================================\n# 6. Test Set Inference and Submission File\n# =============================================================================\n# Collect test files\ntest_dir = \"/kaggle/input/waveform-inversion/test\"\ntest_files = list(Path(test_dir).glob(\"*.npy\"))\nprint(\"Test files:\", len(test_files))\n\n# Define CSV header details\nx_cols = [f\"x_{i}\" for i in range(1, 70, 2)]\ncsv_fields = [\"oid_ypos\"] + x_cols\n\n# Create test DataLoader\ntest_dataset = TestDataset(test_files)\ntest_loader = DataLoader(test_dataset, batch_size=8, num_workers=4, pin_memory=True)\n\n# Switch model to evaluation mode\nmodel.eval()\n\n# Generate submission CSV\nsubmission_path = \"submission.csv\"\nwith open(submission_path, \"wt\", newline=\"\") as csv_file:\n    writer = csv.DictWriter(csv_file, fieldnames=csv_fields)\n    writer.writeheader()\n    for batch_inputs, batch_ids in tqdm(test_loader, desc=\"Test Inference\"):\n        batch_inputs = batch_inputs.to(device)\n        with torch.inference_mode():\n            outputs = model(batch_inputs)\n        # Bring predictions back to CPU as numpy array\n        predictions = outputs[:, 0].cpu().numpy()\n        for pred_map, file_id in zip(predictions, batch_ids):\n            # For each y position, extract every second element from x positions\n            for y_index in range(70):\n                row_dict = {\n                    \"oid_ypos\": f\"{file_id}_y_{y_index}\",\n                    **{x_cols[i]: pred_map[y_index, (i * 2) + 1] for i in range(len(x_cols))}\n                }\n                writer.writerow(row_dict)\n\nprint(f\"Submission file generated: {submission_path}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:22:35.798005Z","iopub.execute_input":"2025-04-12T10:22:35.798438Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Thanks","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}