{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":39763,"databundleVersionId":11756775,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":332421,"sourceType":"modelInstanceVersion","isSourceIdPinned":false,"modelInstanceId":278650,"modelId":299553}],"dockerImageVersionId":30919,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nfrom torch.utils.data import Dataset, DataLoader","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:15:57.208496Z","iopub.execute_input":"2025-04-11T11:15:57.208800Z","iopub.status.idle":"2025-04-11T11:16:01.715614Z","shell.execute_reply.started":"2025-04-11T11:15:57.208769Z","shell.execute_reply":"2025-04-11T11:16:01.714766Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Find files to load and create Dataset","metadata":{}},{"cell_type":"code","source":"all_inputs = [\n    f\n    for f in\n    Path('/kaggle/input/waveform-inversion/train_samples').rglob('*.npy')\n    if ('seis' in f.stem) or ('data' in f.stem)\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:01.716903Z","iopub.execute_input":"2025-04-11T11:16:01.717334Z","iopub.status.idle":"2025-04-11T11:16:01.813361Z","shell.execute_reply.started":"2025-04-11T11:16:01.717300Z","shell.execute_reply":"2025-04-11T11:16:01.812564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def inputs_files_to_output_files(input_files):\n    return [\n        Path(str(f).replace('seis', 'vel').replace('data', 'model'))\n        for f in input_files\n    ]\n\nall_outputs = inputs_files_to_output_files(all_inputs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:01.814545Z","iopub.execute_input":"2025-04-11T11:16:01.814864Z","iopub.status.idle":"2025-04-11T11:16:01.819594Z","shell.execute_reply.started":"2025-04-11T11:16:01.814834Z","shell.execute_reply":"2025-04-11T11:16:01.818678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"assert all(f.exists() for f in all_outputs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:01.820467Z","iopub.execute_input":"2025-04-11T11:16:01.820795Z","iopub.status.idle":"2025-04-11T11:16:01.837384Z","shell.execute_reply.started":"2025-04-11T11:16:01.820762Z","shell.execute_reply":"2025-04-11T11:16:01.836683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_inputs = [all_inputs[i] for i in range(0, len(all_inputs), 2)] # Sample every two\nvalid_inputs = [f for f in all_inputs if not f in train_inputs]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:01.839464Z","iopub.execute_input":"2025-04-11T11:16:01.839828Z","iopub.status.idle":"2025-04-11T11:16:01.853018Z","shell.execute_reply.started":"2025-04-11T11:16:01.839801Z","shell.execute_reply":"2025-04-11T11:16:01.852178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_outputs = inputs_files_to_output_files(train_inputs)\nvalid_outputs = inputs_files_to_output_files(valid_inputs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:01.854364Z","iopub.execute_input":"2025-04-11T11:16:01.854636Z","iopub.status.idle":"2025-04-11T11:16:01.867753Z","shell.execute_reply.started":"2025-04-11T11:16:01.854609Z","shell.execute_reply":"2025-04-11T11:16:01.866883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class SeismicDataset(Dataset):\n    def __init__(self, inputs_files, output_files, n_examples_per_file=500):\n        assert len(inputs_files) == len(output_files)\n        self.inputs_files = inputs_files\n        self.output_files = output_files\n        self.n_examples_per_file = n_examples_per_file\n\n    def __len__(self):\n        return len(self.inputs_files) * self.n_examples_per_file\n\n    def __getitem__(self, idx):\n        # Calculate file offset and sample offset within file\n        file_idx = idx // self.n_examples_per_file\n        sample_idx = idx % self.n_examples_per_file\n\n        X = np.load(self.inputs_files[file_idx], mmap_mode='r')\n        y = np.load(self.output_files[file_idx], mmap_mode='r')\n\n        try:\n            return X[sample_idx].copy(), y[sample_idx].copy()\n        finally:\n            del X, y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:01.868534Z","iopub.execute_input":"2025-04-11T11:16:01.868811Z","iopub.status.idle":"2025-04-11T11:16:01.884376Z","shell.execute_reply.started":"2025-04-11T11:16:01.868782Z","shell.execute_reply":"2025-04-11T11:16:01.883493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dstrain = SeismicDataset(train_inputs, train_outputs)\ndltrain = DataLoader(dstrain, batch_size=128, shuffle=True, pin_memory=True, drop_last=True, num_workers=4, persistent_workers=True)\n\ndsvalid = SeismicDataset(valid_inputs, valid_outputs)\ndlvalid = DataLoader(dsvalid, batch_size=128, shuffle=False, pin_memory=True, drop_last=False, num_workers=4, persistent_workers=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:01.885066Z","iopub.execute_input":"2025-04-11T11:16:01.885364Z","iopub.status.idle":"2025-04-11T11:16:01.900548Z","shell.execute_reply.started":"2025-04-11T11:16:01.885330Z","shell.execute_reply":"2025-04-11T11:16:01.899740Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## DumbNet","metadata":{}},{"cell_type":"code","source":"\nclass ResidualBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, downsample=False):\n        super().__init__()\n        stride = 2 if downsample else 1\n\n        self.conv1 = nn.Conv2d(in_channels, out_channels, kernel_size=3, padding=1, stride=stride)\n        self.bn1 = nn.BatchNorm2d(out_channels)\n        self.relu = nn.ReLU(inplace=True)\n\n        self.conv2 = nn.Conv2d(out_channels, out_channels, kernel_size=3, padding=1)\n        self.bn2 = nn.BatchNorm2d(out_channels)\n\n        self.downsample = None\n        if downsample or in_channels != out_channels:\n            self.downsample = nn.Sequential(\n                nn.Conv2d(in_channels, out_channels, kernel_size=1, stride=stride),\n                nn.BatchNorm2d(out_channels)\n            )\n\n    def forward(self, x):\n        identity = x\n\n        out = self.relu(self.bn1(self.conv1(x)))\n        out = self.bn2(self.conv2(out))\n\n        if self.downsample is not None:\n            identity = self.downsample(identity)\n\n        out += identity\n        return self.relu(out)\n\nclass DumbNet(nn.Module):\n    '''Deep CNN with residual blocks and dense classifier'''\n    def __init__(self, input_channels=5, output_size=70 * 70):\n        super().__init__()\n\n        self.stem = nn.Sequential(\n            nn.Conv2d(input_channels, 64, kernel_size=7, stride=2, padding=3),\n            nn.BatchNorm2d(64),\n            nn.ReLU(inplace=True),\n            nn.MaxPool2d(kernel_size=3, stride=2, padding=1)  # 1000x70 -> ~250x18\n        )\n\n        self.layer1 = ResidualBlock(64, 128, downsample=True)  # ~125x9\n        self.layer2 = ResidualBlock(128, 256, downsample=True)  # ~63x5\n        self.layer3 = ResidualBlock(256, 512, downsample=True)  # ~32x3\n        self.layer4 = ResidualBlock(512, 1024, downsample=True)\n        self.layer5 = ResidualBlock(1024, 1024, downsample = False)# same spatial\n\n        self.global_pool = nn.AdaptiveAvgPool2d((4, 4))  # fixed output\n\n        self.classifier = nn.Sequential(\n            nn.Flatten(),\n            nn.Linear(1024 * 4 * 4, 2048),\n            nn.GELU(),\n            nn.Dropout(0.5),\n\n            nn.Linear(2048, 1024),\n            nn.GELU(),\n            nn.Dropout(0.25),\n\n            nn.Linear(1024, output_size)\n        )\n\n    def forward(self, x):\n        bs = x.size(0)\n\n        x = self.stem(x)\n        x = self.layer1(x)\n        x = self.layer2(x)\n        x = self.layer3(x)\n        x = self.layer4(x)\n        x = self.layer5(x)\n        x = self.global_pool(x)\n\n        x = self.classifier(x)\n        return x.view(bs, 1, 70, 70) * 1000 + 1500","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:01.901562Z","iopub.execute_input":"2025-04-11T11:16:01.901849Z","iopub.status.idle":"2025-04-11T11:16:01.921562Z","shell.execute_reply.started":"2025-04-11T11:16:01.901828Z","shell.execute_reply":"2025-04-11T11:16:01.920580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ndevice","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:01.922810Z","iopub.execute_input":"2025-04-11T11:16:01.923203Z","iopub.status.idle":"2025-04-11T11:16:01.990165Z","shell.execute_reply.started":"2025-04-11T11:16:01.923171Z","shell.execute_reply":"2025-04-11T11:16:01.989465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = DumbNet().to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:01.990988Z","iopub.execute_input":"2025-04-11T11:16:01.991299Z","iopub.status.idle":"2025-04-11T11:16:03.044532Z","shell.execute_reply.started":"2025-04-11T11:16:01.991268Z","shell.execute_reply":"2025-04-11T11:16:03.043909Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Predict test","metadata":{}},{"cell_type":"code","source":"import csv  # Use \"low-level\" CSV to save memory on predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:03.045253Z","iopub.execute_input":"2025-04-11T11:16:03.045474Z","iopub.status.idle":"2025-04-11T11:16:03.049050Z","shell.execute_reply.started":"2025-04-11T11:16:03.045455Z","shell.execute_reply":"2025-04-11T11:16:03.048237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\ntest_files = list(Path('/kaggle/input/waveform-inversion/test').glob('*.npy'))\nlen(test_files)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:03.049714Z","iopub.execute_input":"2025-04-11T11:16:03.049988Z","iopub.status.idle":"2025-04-11T11:16:03.782505Z","shell.execute_reply.started":"2025-04-11T11:16:03.049956Z","shell.execute_reply":"2025-04-11T11:16:03.781865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_cols = [f'x_{i}' for i in range(1, 70, 2)]\nfieldnames = ['oid_ypos'] + x_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:03.784870Z","iopub.execute_input":"2025-04-11T11:16:03.785074Z","iopub.status.idle":"2025-04-11T11:16:03.788652Z","shell.execute_reply.started":"2025-04-11T11:16:03.785056Z","shell.execute_reply":"2025-04-11T11:16:03.787845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TestDataset(Dataset):\n    def __init__(self, test_files):\n        self.test_files = test_files\n\n\n    def __len__(self):\n        return len(self.test_files)\n\n\n    def __getitem__(self, i):\n        test_file = self.test_files[i]\n\n        return np.load(test_file), test_file.stem","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:03.789613Z","iopub.execute_input":"2025-04-11T11:16:03.789896Z","iopub.status.idle":"2025-04-11T11:16:03.805682Z","shell.execute_reply.started":"2025-04-11T11:16:03.789854Z","shell.execute_reply":"2025-04-11T11:16:03.805047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ds = TestDataset(test_files)\ndl = DataLoader(ds, batch_size=128, num_workers=4, pin_memory=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:03.806441Z","iopub.execute_input":"2025-04-11T11:16:03.806739Z","iopub.status.idle":"2025-04-11T11:16:03.821191Z","shell.execute_reply.started":"2025-04-11T11:16:03.806712Z","shell.execute_reply":"2025-04-11T11:16:03.820590Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PATH = \"/kaggle/input/dumbernet/pytorch/default/1/model_84.pth\" \nmodel.eval()\nmodel.load_state_dict(torch.load(PATH, weights_only=True))\n\nwith open('submission.csv', 'wt', newline='') as csvfile:\n    writer = csv.DictWriter(csvfile, fieldnames=fieldnames)\n    writer.writeheader()\n    \n    for inputs, oids_test in tqdm(dl, desc='test'):\n        inputs = inputs.to(device)\n        with torch.inference_mode():\n            outputs = model(inputs)\n\n        y_preds = outputs[:, 0].cpu().numpy()\n        \n        for y_pred, oid_test in zip(y_preds, oids_test):\n            for y_pos in range(70):\n                row = dict(\n                    zip(\n                        x_cols,\n                        [y_pred[y_pos, x_pos] for x_pos in range(1, 70, 2)]\n                    )\n                )\n                row['oid_ypos'] = f\"{oid_test}_y_{y_pos}\"\n            \n                writer.writerow(row)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-11T11:16:03.822018Z","iopub.execute_input":"2025-04-11T11:16:03.822213Z","execution_failed":"2025-04-11T11:16:51.973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}