{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39763,"databundleVersionId":11756775,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":233737401,"sourceType":"kernelVersion"},{"sourceId":233737815,"sourceType":"kernelVersion"},{"sourceId":233738664,"sourceType":"kernelVersion"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:57:17.385876Z","iopub.execute_input":"2025-04-19T14:57:17.386309Z","iopub.status.idle":"2025-04-19T14:58:23.911837Z","shell.execute_reply.started":"2025-04-19T14:57:17.386236Z","shell.execute_reply":"2025-04-19T14:58:23.910792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport glob\nimport os\nfrom tqdm.auto import tqdm\nfrom torch.utils.data import Dataset, DataLoader\nfrom pathlib import Path\nimport csv\nimport pathlib, html\nfrom IPython.display import HTML\nimport utils\nimport network\nimport transforms as T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:23.913578Z","iopub.execute_input":"2025-04-19T14:58:23.913912Z","iopub.status.idle":"2025-04-19T14:58:23.920627Z","shell.execute_reply.started":"2025-04-19T14:58:23.913869Z","shell.execute_reply":"2025-04-19T14:58:23.919404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Root Paths\nBASE_DIR = '/kaggle/input/waveform-inversion'\nTRAIN_DIR = os.path.join(BASE_DIR, 'train_samples')\nTEST_DIR = os.path.join(BASE_DIR, 'test')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:23.921776Z","iopub.execute_input":"2025-04-19T14:58:23.922468Z","iopub.status.idle":"2025-04-19T14:58:23.943403Z","shell.execute_reply.started":"2025-04-19T14:58:23.922445Z","shell.execute_reply":"2025-04-19T14:58:23.942233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def collect_input_files(\n    data_dir: str = TRAIN_DIR\n) -> list:\n    \"\"\"\n    Recursively search for .npy files in data_dir that contain 'seis' or 'data' in their filename.\n\n    Parameters:\n    ----------\n    data_dir : str, default = TRAIN_DIR\n        The data_dir that need searching.\n\n    Returns:\n    -------\n    list\n        A list contains the paths of all input files in our data.\n    \"\"\"\n    return [f for f in Path(data_dir).rglob(\"*.npy\") if (\"seis\" in f.stem) or (\"data\" in f.stem)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:23.945179Z","iopub.execute_input":"2025-04-19T14:58:23.945460Z","iopub.status.idle":"2025-04-19T14:58:23.954531Z","shell.execute_reply.started":"2025-04-19T14:58:23.945438Z","shell.execute_reply":"2025-04-19T14:58:23.953576Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def map_input_to_output(\n    input_files: list\n) -> list:\n    \"\"\"\n    Map each input file to its corresponding output file by replacing keywords.\n\n    Parameters:\n    ----------\n    input_files : list\n        The list that contains the paths of all input files in our data.\n\n    Returns:\n    -------\n    list\n        A list contains the paths of all output files in our data.\n    \"\"\"\n    return [Path(str(f).replace(\"seis\", \"vel\").replace(\"data\", \"model\")) for f in input_files]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:23.955418Z","iopub.execute_input":"2025-04-19T14:58:23.955636Z","iopub.status.idle":"2025-04-19T14:58:23.972069Z","shell.execute_reply.started":"2025-04-19T14:58:23.955620Z","shell.execute_reply":"2025-04-19T14:58:23.971041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_model = np.load('/kaggle/input/waveform-inversion/train_samples/FlatFault_A/seis4_1_0.npy')\nrandom_model.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:23.973192Z","iopub.execute_input":"2025-04-19T14:58:23.973524Z","iopub.status.idle":"2025-04-19T14:58:24.397886Z","shell.execute_reply.started":"2025-04-19T14:58:23.973501Z","shell.execute_reply":"2025-04-19T14:58:24.397000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"batch_size : \", random_model.shape[0])\nprint(\"num_sources, : \", random_model.shape[1])\nprint(\"time_steps : \", random_model.shape[2])\nprint('num_receivers: ',random_model.shape[3])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:24.399026Z","iopub.execute_input":"2025-04-19T14:58:24.399333Z","iopub.status.idle":"2025-04-19T14:58:24.405060Z","shell.execute_reply.started":"2025-04-19T14:58:24.399305Z","shell.execute_reply":"2025-04-19T14:58:24.404069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random_velocity = np.load('/kaggle/input/waveform-inversion/train_samples/FlatFault_A/vel4_1_0.npy')\nrandom_velocity.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:24.405969Z","iopub.execute_input":"2025-04-19T14:58:24.406237Z","iopub.status.idle":"2025-04-19T14:58:24.432207Z","shell.execute_reply.started":"2025-04-19T14:58:24.406218Z","shell.execute_reply":"2025-04-19T14:58:24.431422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"batch_size : \", random_velocity.shape[0])\nprint(\"num_sources, : \", random_velocity.shape[1])\nprint(\"height : \", random_velocity.shape[2])\nprint('width: ',random_velocity.shape[3])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:24.433082Z","iopub.execute_input":"2025-04-19T14:58:24.433314Z","iopub.status.idle":"2025-04-19T14:58:24.438646Z","shell.execute_reply.started":"2025-04-19T14:58:24.433297Z","shell.execute_reply":"2025-04-19T14:58:24.437807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT = next(pathlib.Path(\"/kaggle/input\").rglob(\"train_samples\"), None)\nassert ROOT and ROOT.is_dir(), \"❌  /train_samples not found\"\n\n# ---------- helpers ----------\ndef human_bytes(n):\n    units = ['B','KB','MB','GB','TB']; i = 0\n    while n >= 1024 and i < len(units)-1: n /= 1024; i += 1\n    return f\"{n:.1f}{units[i]}\"\n\ndef npy_info(p):\n    arr = np.load(p, mmap_mode='r'); shp, dt = arr.shape, arr.dtype; arr._mmap.close()\n    return shp, dt\n\ndef folder_desc(name):\n    if name.startswith(\"FlatVel\"):   base = \"Vel—flat, gently layered\"; \n    elif name.startswith(\"CurveVel\"): base = \"Vel—curved/folded layers\";\n    elif name.startswith(\"FlatFault\"): base = \"Fault—flat layers with breaks\";\n    elif name.startswith(\"CurveFault\"):base = \"Fault—curved layers with breaks\";\n    elif name.startswith(\"Style\"):   base = \"Style—random texture pattern\";\n    else:                             base = \"Unknown pattern\"\n    level = \"A simpler\" if name.endswith(\"_A\") else \"B more complex\" if name.endswith(\"_B\") else \"\"\n    return f\"{base} ({level})\".strip()\n\nKIND_INFO = {\n    \"Data\":  \"4‑D seismic waveforms (sources × time × receivers)\",\n    \"Model\": \"2‑D velocity ground truth\",\n    \"Seis\":  \"Seismic recordings (prefix layout)\",\n    \"Vel\":   \"Velocity maps (prefix layout)\",\n    \"Files\": \"Misc. .npy files\"\n}\n\n# ---------- gather ----------\ntree = {}\nfor fld in sorted(ROOT.iterdir()):\n    if not fld.is_dir(): continue\n    kinds={}\n    def add(k, paths):\n        items=[(p.name, f\"shape={shp}, dtype={dt}, size={human_bytes(p.stat().st_size)}\")\n               for p in paths if (shp:=npy_info(p)[0]) or True for dt in [npy_info(p)[1]]]\n        if items: kinds[k]={\"count\":len(items),\"files\":items}\n    add(\"Data\",  (fld/\"data\").glob(\"*.npy\")  if (fld/\"data\").is_dir()  else [])\n    add(\"Model\", (fld/\"model\").glob(\"*.npy\") if (fld/\"model\").is_dir() else [])\n    add(\"Seis\", fld.glob(\"seis*.npy\"));  add(\"Vel\", fld.glob(\"vel*.npy\"))\n    if not kinds: add(\"Files\", fld.glob(\"*.npy\"))\n    tree[fld.name]={\"desc\":folder_desc(fld.name),\"kinds\":kinds}\n\n# ---------- html ----------\nhtml_parts=[\"\"\"\n<style>\n:root{--accent:#136efd;--bg:#fafafa;--border:#ddd;--font:system-ui,sans-serif}\n#exp{font-family:var(--font);background:var(--bg);padding:1rem;border:1px solid var(--border);\n     border-radius:8px;max-width:1050px;margin:auto}\n#exp h2{margin:0 0 1rem;text-align:center;font-size:1.35rem}\n.folder{border:1px solid var(--border);border-radius:6px;margin:.7rem 0;background:#fff}\n.folder summary{cursor:pointer;display:flex;gap:.6rem;align-items:center;padding:.5rem .75rem}\n.fname{font-weight:600}\n.fdesc{font-size:.8rem;color:#555}\n.kind-line{margin-left:1.3rem;margin-top:.45rem}\n.kind-tag{background:var(--accent);color:#fff;border-radius:4px;padding:.18rem .55rem;font-size:.75rem}\n.kind-info{font-size:.78rem;color:#222;margin-left:.45rem}\n.count{color:#555;font-size:.78rem;margin-left:.25rem}\n.file-list{margin:.25rem 0 .8rem 2.5rem;font-size:.85rem;color:#333}\n.file-list li{margin:.04rem 0;list-style-type:disc}\n</style>\n<div id=\"exp\">\n  <h2>Training .npy Explorer — Inline Explanations</h2>\n\"\"\"]\n\nfor name,info in tree.items():\n    kinds=info[\"kinds\"]; desc=html.escape(info[\"desc\"])\n    html_parts.append(\"<details class='folder'>\")\n    html_parts.append(f\"<summary><span class='fname'>{html.escape(name)}</span>\"\n                      f\"<span class='fdesc'>— {desc}</span></summary>\")\n    for kind,meta in kinds.items():\n        html_parts.append(f\"<div class='kind-line'><span class='kind-tag'>{kind}</span>\"\n                          f\"<span class='count'>({meta['count']})</span>\"\n                          f\"<span class='kind-info'>{html.escape(KIND_INFO.get(kind,''))}</span></div>\")\n        html_parts.append(\"<ul class='file-list'>\")\n        for fn,tip in meta[\"files\"]:\n            html_parts.append(f\"<li title='{html.escape(tip)}'>{html.escape(fn)}</li>\")\n        html_parts.append(\"</ul>\")\n    html_parts.append(\"</details>\")\n\nhtml_parts.append(\"</div>\")\nHTML(\"\".join(html_parts))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:24.441068Z","iopub.execute_input":"2025-04-19T14:58:24.441395Z","iopub.status.idle":"2025-04-19T14:58:24.698513Z","shell.execute_reply.started":"2025-04-19T14:58:24.441367Z","shell.execute_reply":"2025-04-19T14:58:24.697562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size, num_sources, time_steps, num_receivers = random_model.shape\nprint(f\"Data form: {random_model.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:24.699755Z","iopub.execute_input":"2025-04-19T14:58:24.700079Z","iopub.status.idle":"2025-04-19T14:58:24.704918Z","shell.execute_reply.started":"2025-04-19T14:58:24.700056Z","shell.execute_reply":"2025-04-19T14:58:24.703872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_batch = 0\n\nfig, axes = plt.subplots(nrows=num_sources, figsize=(20, 25), sharex=True, sharey=True)\nfor source in range(num_sources):\n    data = random_model[selected_batch, source, :, :].T\n    im = axes[source].imshow(\n        data,\n        cmap='seismic',\n        aspect='auto',\n        extent=[0, time_steps, num_receivers, 0],\n        vmin=-np.abs(data).max(),\n        vmax=np.abs(data).max()\n    )\n    axes[source].set_title(f'Source {source}', pad=15)\n    axes[source].set_ylabel('Receiver number')\n    axes[source].set_xlabel('Time step' if source == num_sources-1 else '')\nplt.tight_layout()\nplt.show();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:24.705923Z","iopub.execute_input":"2025-04-19T14:58:24.706337Z","iopub.status.idle":"2025-04-19T14:58:26.764626Z","shell.execute_reply.started":"2025-04-19T14:58:24.706308Z","shell.execute_reply":"2025-04-19T14:58:26.763781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_batch = 0\nselected_source = 2\n\ndata = random_model[selected_batch, selected_source, :, :]\nX, Y = np.meshgrid(np.arange(num_receivers), np.arange(time_steps))\nfig = plt.figure(figsize=(18, 10))\nax = fig.add_subplot(111, projection='3d')\nsurf = ax.plot_surface(X, Y, data,cmap='seismic', rstride=1, cstride=1)\nax.set_title(f'3D visualization of seismic data (batch={selected_batch}, source={selected_source})')\nax.set_xlabel('Receiver number')\nax.set_ylabel('Time step')\nax.set_zlabel('The amplitude')\nfig.colorbar(surf, ax=ax, shrink=0.5, label='The amplitude')\nplt.show();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:26.766068Z","iopub.execute_input":"2025-04-19T14:58:26.766384Z","iopub.status.idle":"2025-04-19T14:58:32.000670Z","shell.execute_reply.started":"2025-04-19T14:58:26.766359Z","shell.execute_reply":"2025-04-19T14:58:31.999734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class SeismicDataset(Dataset):\n    \"\"\"\n    Dataset handling seismic files with multiple examples per file.\n    \"\"\"\n    def __init__(self, in_files: list, out_files: list, examples_per_file: int = 500):\n        assert len(in_files) == len(out_files)\n        self.in_files = in_files\n        self.out_files = out_files\n        self.examples_per_file = examples_per_file\n\n    def __len__(self):\n        return len(self.in_files) * self.examples_per_file\n\n    def __getitem__(self, idx: int):\n        file_index = idx // self.examples_per_file\n        sample_index = idx % self.examples_per_file\n\n        # Memory map the file to reduce memory usage\n        x_data = np.load(self.in_files[file_index], mmap_mode=\"r\")\n        y_data = np.load(self.out_files[file_index], mmap_mode=\"r\")\n        try:\n            return x_data[sample_index].copy(), y_data[sample_index].copy()\n        finally:\n            del x_data, y_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:32.001784Z","iopub.execute_input":"2025-04-19T14:58:32.002425Z","iopub.status.idle":"2025-04-19T14:58:32.011545Z","shell.execute_reply.started":"2025-04-19T14:58:32.002391Z","shell.execute_reply":"2025-04-19T14:58:32.010602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inputs_all = collect_input_files(TRAIN_DIR)\noutputs_all = map_input_to_output(inputs_all)\n\n# Check all output files exist\nassert all(f.exists() for f in outputs_all)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:32.012722Z","iopub.execute_input":"2025-04-19T14:58:32.013056Z","iopub.status.idle":"2025-04-19T14:58:32.055436Z","shell.execute_reply.started":"2025-04-19T14:58:32.013025Z","shell.execute_reply":"2025-04-19T14:58:32.054474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_inputs = [inputs_all[i] for i in range(0, len(inputs_all), 2)]\nvalid_inputs = [f for f in inputs_all if f not in train_inputs]\ntrain_outputs = map_input_to_output(train_inputs)\nvalid_outputs = map_input_to_output(valid_inputs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:32.056400Z","iopub.execute_input":"2025-04-19T14:58:32.056635Z","iopub.status.idle":"2025-04-19T14:58:32.062147Z","shell.execute_reply.started":"2025-04-19T14:58:32.056617Z","shell.execute_reply":"2025-04-19T14:58:32.061112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = SeismicDataset(train_inputs, train_outputs, examples_per_file=500)\nvalid_dataset = SeismicDataset(valid_inputs, valid_outputs, examples_per_file=500)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:32.063112Z","iopub.execute_input":"2025-04-19T14:58:32.063416Z","iopub.status.idle":"2025-04-19T14:58:32.082124Z","shell.execute_reply.started":"2025-04-19T14:58:32.063393Z","shell.execute_reply":"2025-04-19T14:58:32.081104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_loader = DataLoader(\n    train_dataset, # Dataset from which to load the data.\n    batch_size=64, # How many samples per batch to load.\n    shuffle=True, # Set to True to have the data reshuffled at every epoch.\n    pin_memory=True, # Copy Tensors into device/CUDA pinned memory before returning them. \n    drop_last=True, # Set to True to drop the last incomplete batch.\n    num_workers=4, # How many subprocesses to use for data loading.\n    persistent_workers=True # Will not shutdown the worker processes after a dataset has been consumed once.\n)\n\nvalid_loader = DataLoader(\n    valid_dataset, # Dataset from which to load the data.\n    batch_size=64, # How many samples per batch to load.\n    shuffle=True, # Set to True to have the data reshuffled at every epoch\n    pin_memory=True, # Copy Tensors into device/CUDA pinned memory before returning them.\n    drop_last=True, # Set to True to drop the last incomplete batch.\n    num_workers=4, # How many subprocesses to use for data loading.\n    persistent_workers=True # Will not shutdown the worker processes after a dataset has been consumed once.\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:32.083282Z","iopub.execute_input":"2025-04-19T14:58:32.083731Z","iopub.status.idle":"2025-04-19T14:58:32.219254Z","shell.execute_reply.started":"2025-04-19T14:58:32.083699Z","shell.execute_reply":"2025-04-19T14:58:32.218616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_features, train_labels = next(iter(train_loader))\nprint(f\"Feature batch shape: {train_features.size()}\")\nprint(f\"Labels batch shape: {train_labels.size()}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T14:58:32.220153Z","iopub.execute_input":"2025-04-19T14:58:32.220445Z","iopub.status.idle":"2025-04-19T14:58:34.014824Z","shell.execute_reply.started":"2025-04-19T14:58:32.220416Z","shell.execute_reply":"2025-04-19T14:58:34.012420Z"}},"outputs":[],"execution_count":null}]}