{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":39763,"databundleVersionId":11756775,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:08.621399Z","iopub.execute_input":"2025-04-14T07:37:08.621636Z","iopub.status.idle":"2025-04-14T07:37:08.973289Z","shell.execute_reply.started":"2025-04-14T07:37:08.621613Z","shell.execute_reply":"2025-04-14T07:37:08.972407Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Data and Labels","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\n# Example for loading seismic data and corresponding velocity map\nseismic_data = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/data/data1.npy')\nvelocity_map = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/model/model1.npy')\n\n# You can load other samples similarly\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:08.975199Z","iopub.execute_input":"2025-04-14T07:37:08.975467Z","iopub.status.idle":"2025-04-14T07:37:09.205790Z","shell.execute_reply.started":"2025-04-14T07:37:08.975452Z","shell.execute_reply":"2025-04-14T07:37:09.205029Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Option 1: Check Available Files","metadata":{}},{"cell_type":"code","source":"import os\n\ndata_dir = '/kaggle/input/waveform-inversion/train_samples/FlatVel_A/data/'\nmodel_dir = '/kaggle/input/waveform-inversion/train_samples/FlatVel_A/model/'\n\n# List the files in the data and model directories\ndata_files = os.listdir(data_dir)\nmodel_files = os.listdir(model_dir)\n\nprint(f\"Available data files: {data_files}\")\nprint(f\"Available model files: {model_files}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:09.206551Z","iopub.execute_input":"2025-04-14T07:37:09.206760Z","iopub.status.idle":"2025-04-14T07:37:09.212685Z","shell.execute_reply.started":"2025-04-14T07:37:09.206742Z","shell.execute_reply":"2025-04-14T07:37:09.211912Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Option 2: Dynamically Load Existing Files","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport os\n\nX = []  # Store seismic data\ny = []  # Store velocity maps (labels)\n\n# Directory paths\ndata_dir = '/kaggle/input/waveform-inversion/train_samples/FlatVel_A/data/'\nmodel_dir = '/kaggle/input/waveform-inversion/train_samples/FlatVel_A/model/'\n\n# Loop through available files and load them\ndata_files = sorted(os.listdir(data_dir))\nmodel_files = sorted(os.listdir(model_dir))\n\n# Ensure the files match and we have a corresponding data and model pair\nfor data_file, model_file in zip(data_files, model_files):\n    if data_file.endswith('.npy') and model_file.endswith('.npy'):\n        seismic_data = np.load(os.path.join(data_dir, data_file))\n        velocity_map = np.load(os.path.join(model_dir, model_file))\n        \n        X.append(seismic_data)\n        y.append(velocity_map)\n\n# Convert to numpy arrays\nX = np.array(X)\ny = np.array(y)\n\nprint(f\"Seismic Data Shape: {X.shape}\")\nprint(f\"Velocity Maps Shape: {y.shape}\")\n\n# Proceed to split the dataset\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\nprint(f\"X_train shape: {X_train.shape}, X_val shape: {X_val.shape}\")\nprint(f\"y_train shape: {y_train.shape}, y_val shape: {y_val.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:09.213565Z","iopub.execute_input":"2025-04-14T07:37:09.213836Z","iopub.status.idle":"2025-04-14T07:37:11.130402Z","shell.execute_reply.started":"2025-04-14T07:37:09.213813Z","shell.execute_reply":"2025-04-14T07:37:11.129778Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Prepare the Train and Validation Data","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\ndata = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/data/data1.npy')\nmodel = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/model/model1.npy')\n\nprint(\"Data shape:\", data.shape)    # e.g., (500, 32, 1000, 100)\nprint(\"Model shape:\", model.shape)  # e.g., (500, 100, 70)\n\n#plt.imshow(model[0], cmap='jet')\n#plt.imshow(model[0].squeeze(), cmap='jet')\nplt.imshow(model[0][0], cmap='jet')\n\nplt.colorbar(label='Velocity (m/s)')\nplt.title(\"Sample velocity map\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:11.131143Z","iopub.execute_input":"2025-04-14T07:37:11.131505Z","iopub.status.idle":"2025-04-14T07:37:11.642417Z","shell.execute_reply.started":"2025-04-14T07:37:11.131477Z","shell.execute_reply":"2025-04-14T07:37:11.641782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/data/data1.npy')\nprint(train_data.shape)  # Check the shape of the original data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:11.643193Z","iopub.execute_input":"2025-04-14T07:37:11.643541Z","iopub.status.idle":"2025-04-14T07:37:11.859195Z","shell.execute_reply.started":"2025-04-14T07:37:11.643518Z","shell.execute_reply":"2025-04-14T07:37:11.858535Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Option 1: Reshaping to (500, 1000, 70, 5)","metadata":{}},{"cell_type":"code","source":"train_data = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/data/data1.npy')\nprint(train_data.shape)  # Output should be (500, 5, 1000, 70)\n\n# Reshape it to (500, 1000, 70, 5)\ntrain_data_reshaped = train_data.transpose(0, 2, 3, 1)  # (500, 1000, 70, 5)\nprint(train_data_reshaped.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:11.859924Z","iopub.execute_input":"2025-04-14T07:37:11.860228Z","iopub.status.idle":"2025-04-14T07:37:12.078159Z","shell.execute_reply.started":"2025-04-14T07:37:11.860207Z","shell.execute_reply":"2025-04-14T07:37:12.077507Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Option 2: Flattening the source dimension into the last axis (if needed for your model)","metadata":{}},{"cell_type":"code","source":"train_data = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/data/data1.npy')\nprint(train_data.shape)  # Output should be (500, 5, 1000, 70)\n\n# Flatten the sources dimension\ntrain_data_reshaped = train_data.reshape(500, 1000, 70 * 5)  # Flatten sources (5) into the last dimension\nprint(train_data_reshaped.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:12.080253Z","iopub.execute_input":"2025-04-14T07:37:12.080484Z","iopub.status.idle":"2025-04-14T07:37:12.298620Z","shell.execute_reply.started":"2025-04-14T07:37:12.080464Z","shell.execute_reply":"2025-04-14T07:37:12.297990Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Building a 2D CNN Model for Option 1","metadata":{"_kg_hide-input":true}},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense\n\n# Build the CNN model\nmodel = Sequential([\n    # Convolutional layer with 32 filters, kernel size (3, 3), and input shape (1000, 70, 5)\n    Conv2D(32, (3, 3), activation='relu', input_shape=(1000, 70, 5)),\n    MaxPooling2D((2, 2)),\n    \n    # Add another convolutional layer\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    \n    # Flatten the output from the convolutional layers\n    Flatten(),\n    \n    # Dense layer for classification or regression\n    Dense(128, activation='relu'),\n    Dense(1)  # Or change the number of units based on your output requirements\n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='mse')  # Change the loss function depending on the task\n\n# Summary of the model\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:12.299312Z","iopub.execute_input":"2025-04-14T07:37:12.299571Z","iopub.status.idle":"2025-04-14T07:37:16.434943Z","shell.execute_reply.started":"2025-04-14T07:37:12.299548Z","shell.execute_reply":"2025-04-14T07:37:16.434385Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Building a 2D CNN Model for Option 2","metadata":{}},{"cell_type":"code","source":"# Build the CNN model for the flattened source-receiver case\nmodel = Sequential([\n    # Convolutional layer with 32 filters, kernel size (3, 3), and input shape (1000, 350)\n    Conv2D(32, (3, 3), activation='relu', input_shape=(1000, 350, 1)),\n    MaxPooling2D((2, 2)),\n    \n    # Add another convolutional layer\n    Conv2D(64, (3, 3), activation='relu'),\n    MaxPooling2D((2, 2)),\n    \n    # Flatten the output from the convolutional layers\n    Flatten(),\n    \n    # Dense layer for classification or regression\n    Dense(128, activation='relu'),\n    Dense(1)  # Or change the number of units based on your output requirements\n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='mse')  # Change the loss function depending on the task\n\n# Summary of the model\nmodel.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:16.435680Z","iopub.execute_input":"2025-04-14T07:37:16.436237Z","iopub.status.idle":"2025-04-14T07:37:16.583811Z","shell.execute_reply.started":"2025-04-14T07:37:16.436213Z","shell.execute_reply":"2025-04-14T07:37:16.583087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# Load the seismic data (X_train)\nX_train = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/data/data1.npy')\n\n# Load the corresponding velocity map data (y_train)\ny_train = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/model/model1.npy')\n\n# Check shapes\nprint(f\"X_train shape: {X_train.shape}\")\nprint(f\"y_train shape: {y_train.shape}\")\n\n# Optionally, load and check validation set (X_val, y_val)\nX_val = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/data/data2.npy')\ny_val = np.load('/kaggle/input/waveform-inversion/train_samples/FlatVel_A/model/model2.npy')\n\nprint(f\"X_val shape: {X_val.shape}\")\nprint(f\"y_val shape: {y_val.shape}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:16.584602Z","iopub.execute_input":"2025-04-14T07:37:16.584856Z","iopub.status.idle":"2025-04-14T07:37:17.020604Z","shell.execute_reply.started":"2025-04-14T07:37:16.584840Z","shell.execute_reply":"2025-04-14T07:37:17.020000Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Setup","metadata":{}},{"cell_type":"code","source":"model.layers  # list of layers\nmodel.summary()  # prints model architecture","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:17.021331Z","iopub.execute_input":"2025-04-14T07:37:17.021547Z","iopub.status.idle":"2025-04-14T07:37:17.035981Z","shell.execute_reply.started":"2025-04-14T07:37:17.021530Z","shell.execute_reply":"2025-04-14T07:37:17.035420Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Before training, reshape X_train and X_val like this","metadata":{}},{"cell_type":"code","source":"# First, check the shape\nprint(X_train.shape)  # Likely (500, 5, 1000, 70, 1, 1, 1, 1, 1, 1, 1)\n\n# Remove extra singleton dimensions\nX_train = np.squeeze(X_train)\nX_val = np.squeeze(X_val)\n\n# Then expand ONLY the last channel dimension\nX_train = np.expand_dims(X_train, axis=-1)  # Now shape: (500, 5, 1000, 70, 1)\nX_val = np.expand_dims(X_val, axis=-1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:17.037060Z","iopub.execute_input":"2025-04-14T07:37:17.037314Z","iopub.status.idle":"2025-04-14T07:37:17.043744Z","shell.execute_reply.started":"2025-04-14T07:37:17.037298Z","shell.execute_reply":"2025-04-14T07:37:17.043080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train = np.squeeze(y_train)\ny_val = np.squeeze(y_val)\ny_train = np.expand_dims(y_train, axis=-1)  # (500, 70, 70, 1)\ny_val = np.expand_dims(y_val, axis=-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:17.044381Z","iopub.execute_input":"2025-04-14T07:37:17.044571Z","iopub.status.idle":"2025-04-14T07:37:17.058723Z","shell.execute_reply.started":"2025-04-14T07:37:17.044556Z","shell.execute_reply":"2025-04-14T07:37:17.057976Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Before training:","metadata":{}},{"cell_type":"code","source":"# Remove all singleton dimensions\nX_train = np.squeeze(X_train)\nX_val = np.squeeze(X_val)\n\n# Add back a single channel dimension at the end (for Conv3D)\nX_train = np.expand_dims(X_train, axis=-1)  # shape: (500, 5, 1000, 70, 1)\nX_val = np.expand_dims(X_val, axis=-1)\n\n# Do the same for y\ny_train = np.squeeze(y_train)\ny_val = np.squeeze(y_val)\n\ny_train = np.expand_dims(y_train, axis=-1)  # shape: (500, 70, 70, 1)\ny_val = np.expand_dims(y_val, axis=-1)\n\n# Print to verify\nprint(\"X_train:\", X_train.shape)\nprint(\"y_train:\", y_train.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:17.059467Z","iopub.execute_input":"2025-04-14T07:37:17.060083Z","iopub.status.idle":"2025-04-14T07:37:17.075473Z","shell.execute_reply.started":"2025-04-14T07:37:17.060058Z","shell.execute_reply":"2025-04-14T07:37:17.074819Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Tour 3D CNN model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import layers, models\n\nmodel = models.Sequential([\n    layers.Conv3D(32, (3, 3, 3), activation='relu', input_shape=X_train.shape[1:], padding='same'),\n    layers.MaxPooling3D((2, 2, 2)),\n    layers.Conv3D(64, (3, 3, 3), activation='relu', padding='same'),\n    layers.MaxPooling3D((2, 2, 2)),\n    layers.Flatten(),\n    layers.Dense(256, activation='relu'),\n    layers.Dense(np.prod(y_train.shape[1:]), activation='linear'),\n    layers.Reshape(y_train.shape[1:])\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:17.076226Z","iopub.execute_input":"2025-04-14T07:37:17.076448Z","iopub.status.idle":"2025-04-14T07:37:17.129968Z","shell.execute_reply.started":"2025-04-14T07:37:17.076425Z","shell.execute_reply":"2025-04-14T07:37:17.129189Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Compile the Model","metadata":{}},{"cell_type":"code","source":"model.compile(optimizer='adam', loss='mse', metrics=['mae'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:17.130777Z","iopub.execute_input":"2025-04-14T07:37:17.131033Z","iopub.status.idle":"2025-04-14T07:37:17.138630Z","shell.execute_reply.started":"2025-04-14T07:37:17.131007Z","shell.execute_reply":"2025-04-14T07:37:17.137959Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train the Model","metadata":{}},{"cell_type":"code","source":"history = model.fit(\n    X_train, y_train,\n    validation_data=(X_val, y_val),\n    epochs=10,\n    batch_size=8,\n    verbose=1\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:37:17.139403Z","iopub.execute_input":"2025-04-14T07:37:17.139656Z","iopub.status.idle":"2025-04-14T07:38:51.241707Z","shell.execute_reply.started":"2025-04-14T07:37:17.139628Z","shell.execute_reply":"2025-04-14T07:38:51.241096Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Plot the Loss Curve ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Val Loss')\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"MSE Loss\")\nplt.legend()\nplt.title(\"Training Curve\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:38:51.242935Z","iopub.execute_input":"2025-04-14T07:38:51.243355Z","iopub.status.idle":"2025-04-14T07:38:51.411152Z","shell.execute_reply.started":"2025-04-14T07:38:51.243329Z","shell.execute_reply":"2025-04-14T07:38:51.410409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize Predictions vs. Ground Truth","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Pick a few samples\nnum_samples = 5\nindices = np.random.choice(len(X_val), num_samples, replace=False)\n\nfor idx in indices:\n    pred = model.predict(X_val[idx:idx+1])[0, 0]\n    true = y_val[idx, 0]\n\n    fig, axes = plt.subplots(1, 2, figsize=(10, 4))\n    axes[0].imshow(true, cmap='jet')\n    axes[0].set_title(\"Ground Truth\")\n\n    axes[1].imshow(pred, cmap='jet')\n    axes[1].set_title(\"Prediction\")\n\n    plt.suptitle(f\"Sample {idx}\")\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:38:51.411854Z","iopub.execute_input":"2025-04-14T07:38:51.412065Z","iopub.status.idle":"2025-04-14T07:38:53.805405Z","shell.execute_reply.started":"2025-04-14T07:38:51.412034Z","shell.execute_reply":"2025-04-14T07:38:53.804785Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Export Predictions","metadata":{"_kg_hide-output":true}},{"cell_type":"code","source":"preds = model.predict(X_val)\nnp.save(\"predicted_velocity_maps.npy\", preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-14T07:38:53.806102Z","iopub.execute_input":"2025-04-14T07:38:53.806315Z","iopub.status.idle":"2025-04-14T07:39:00.372818Z","shell.execute_reply.started":"2025-04-14T07:38:53.806299Z","shell.execute_reply":"2025-04-14T07:39:00.372204Z"}},"outputs":[],"execution_count":null}]}