{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39763,"databundleVersionId":11756775,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -*- coding: utf-8 -*-\n\"\"\"\nGeophysical Waveform Inversion - Submission Notebook\n\nThis notebook provides a basic structure for generating a submission file\nfor the Yale/UNC-CH Geophysical Waveform Inversion competition on Kaggle.\n\"\"\"\n\n# ## 1. Import Libraries\n\nimport pandas as pd\nimport numpy as np\n# Add other necessary libraries here (e.g., your ML framework)\n\n# ## 2. Define Data Paths (if applicable)\n\n# If your data files are located in specific directories, define those paths here.\n# For example:\n# train_data_path = '../input/your-competition-data/train.csv'\n# test_data_path = '../input/your-competition-data/test.csv'\n# sample_submission_path = '../input/your-competition-data/sample_submission.csv'\n\n# ## 3. Load Data (if needed for preprocessing or understanding)\n\n# You might load the test set or sample submission to understand the structure.\n# try:\n#     test_df = pd.read_csv(test_data_path)\n#     sample_submission_df = pd.read_csv(sample_submission_path)\n#     print(\"Test data and sample submission loaded successfully.\")\n#     print(f\"Test data shape: {test_df.shape}\")\n#     print(f\"Sample submission shape: {sample_submission_df.shape}\")\n# except FileNotFoundError:\n#     print(\"Data files not found. Ensure the paths are correct.\")\n\n# ## 4. Load Your Trained Model(s)\n\n# This is where you would load your trained machine learning model(s).\n# The specific code will depend on the framework you are using (e.g., scikit-learn, TensorFlow, PyTorch).\n\n# Example using a hypothetical model object:\n# loaded_model = load_your_model('path/to/your/model.pkl')\n\n# ## 5. Generate Predictions for the Test Set\n\n# This is the core part where you'll use your loaded model(s) to make predictions\n# for the required odd-numbered x columns for all oid_ypos in the test set.\n\n# --- Placeholder for your prediction logic ---\n# You will need to:\n# 1. Load the test data (or identify the relevant input features).\n# 2. Iterate through the unique 'oid_ypos' in the test data.\n# 3. For each 'oid_ypos', generate predictions for x_1, x_3, ..., x_69.\n# 4. Store these predictions in a suitable data structure (e.g., a dictionary or a list of lists).\n\n# Example (Illustrative - replace with your actual code):\n# predictions = {}\n# unique_oids_ypos = ['000039dca2_y_0', '000039dca2_y_1', '...', ...] # Get this from your test data\n#\n# for oid_ypos in unique_oids_ypos:\n#     # Extract features relevant to this oid_ypos\n#     features = get_features_for(oid_ypos)\n#     # Make predictions for the odd x columns\n#     preds = your_model.predict(features) # Adapt based on your model output\n#     predictions[oid_ypos] = {\n#         'x_1': preds[0],\n#         'x_3': preds[1],\n#         'x_5': preds[2],\n#         # ... and so on for all odd x columns\n#     }\n\n# ## 6. Create the Submission DataFrame\n\n# Now, we'll structure the predictions into the required DataFrame format for submission.\n\n# submission_data = []\n# for oid_ypos, preds in predictions.items():\n#     row = {'oid_ypos': oid_ypos}\n#     row.update(preds)\n#     submission_data.append(row)\n#\n# submission_df = pd.DataFrame(submission_data)\n\n# Ensure the order of columns is correct\n# submission_cols = ['oid_ypos'] + [f'x_{i}' for i in range(1, 70) if i % 2 != 0]\n# submission_df = submission_df[submission_cols]\n\n# ## 7. Save the Submission File\n\n# Finally, save the DataFrame to a CSV file named 'submission.csv'.\n\n# submission_df.to_csv('submission.csv', index=False)\n# print(\"Submission file 'submission.csv' created successfully.\")\n\n# ## 8. (Optional) Verify Submission Format\n\n# You might want to load the generated CSV to quickly check its format.\n# loaded_submission_df = pd.read_csv('submission.csv')\n# print(f\"Generated submission shape: {loaded_submission_df.shape}\")\n# print(loaded_submission_df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# -*- coding: utf-8 -*-\n\"\"\"\nGeophysical Waveform Inversion - Submission Notebook\n\nThis notebook provides a basic structure for generating a submission file\nfor the Yale/UNC-CH Geophysical Waveform Inversion competition on Kaggle.\n\"\"\"\n\n# ## 1. Import Libraries\n\nimport pandas as pd\nimport numpy as np\n# Add other necessary libraries here (e.g., your ML framework)\n# Example:\n# from sklearn.ensemble import RandomForestRegressor\nimport joblib\n\n# ## 2. Define Data Paths (if applicable)\n\n# If your data files are located in specific directories, define those paths here.\n# For example:\n# train_data_path = '../input/your-competition-data/train.csv'\n# test_data_path = '../input/your-competition-data/test.csv'\n# sample_submission_path = '../input/your-competition-data/sample_submission.csv'\n\n# For this example, we'll simulate some oid_ypos values\nexample_oids_ypos = [f'000039dca2_y_{i}' for i in range(5)] + [f'123456abc_y_{i}' for i in range(3)]\n\n# ## 3. Load Data (if needed for preprocessing or understanding)\n\n# You might load the test set or sample submission to understand the structure.\n# try:\n#     test_df = pd.read_csv(test_data_path)\n#     sample_submission_df = pd.read_csv(sample_submission_path)\n#     print(\"Test data and sample submission loaded successfully.\")\n#     print(f\"Test data shape: {test_df.shape}\")\n#     print(f\"Sample submission shape: {sample_submission_df.shape}\")\n# except FileNotFoundError:\n#     print(\"Data files not found. Ensure the paths are correct.\")\n\n# ## 4. Load Your Trained Model(s)\n\n# This is where you would load your trained machine learning model(s).\n# The specific code will depend on the framework you are using (e.g., scikit-learn, TensorFlow, PyTorch).\n\n# Example using joblib for a scikit-learn model:\ndef load_your_model(model_path):\n    try:\n        loaded_model = joblib.load(model_path)\n        print(f\"Model loaded successfully from: {model_path}\")\n        return loaded_model\n    except FileNotFoundError:\n        print(f\"Error: Model file not found at {model_path}\")\n        return None\n\n# Replace 'path/to/your/trained_model.pkl' with the actual path to your model file\n# loaded_model = load_your_model('path/to/your/trained_model.pkl')\nloaded_model = None # Placeholder if you don't have a saved model yet\n\n# ## 5. Generate Predictions for the Test Set\n\n# This is the core part where you'll use your loaded model(s) to make predictions\n# for the required odd-numbered x columns for all oid_ypos in the test set.\n\ndef generate_predictions(oids_ypos, model=None):\n    predictions = {}\n    num_odd_x_cols = 35  # x_1, x_3, ..., x_69\n    for oid_ypos in oids_ypos:\n        if model:\n            # Replace this with your actual feature extraction and prediction logic\n            # Example: features = extract_features(oid_ypos)\n            dummy_features = np.random.rand(10) # Example feature vector\n            preds = model.predict(dummy_features.reshape(1, -1))[0] # Adapt based on your model output\n            preds_dict = {}\n            for i in range(num_odd_x_cols):\n                col_index = 2 * i + 1\n                preds_dict[f'x_{col_index}'] = preds[i % len(preds)] # Cycle through predictions if needed\n        else:\n            # Generate dummy predictions if no model is loaded\n            dummy_preds = np.random.rand(num_odd_x_cols) * 3000.0\n            preds_dict = {}\n            for i in range(num_odd_x_cols):\n                col_index = 2 * i + 1\n                preds_dict[f'x_{col_index}'] = dummy_preds[i]\n        predictions[oid_ypos] = preds_dict\n    return predictions\n\n# Replace example_oids_ypos with the actual list from your test data\npredictions = generate_predictions(example_oids_ypos, model=loaded_model)\n\n# ## 6. Create the Submission DataFrame\n\n# Now, we'll structure the predictions into the required DataFrame format for submission.\n\nsubmission_data = []\nfor oid_ypos, preds in predictions.items():\n    row = {'oid_ypos': oid_ypos}\n    row.update(preds)\n    submission_data.append(row)\n\nsubmission_df = pd.DataFrame(submission_data)\n\n# Ensure the order of columns is correct\nsubmission_cols = ['oid_ypos'] + [f'x_{i}' for i in range(1, 70) if i % 2 != 0]\nsubmission_df = submission_df[submission_cols]\n\n# ## 7. Save the Submission File\n\n# Finally, save the DataFrame to a CSV file named 'submission.csv'.\n\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"Submission file 'submission.csv' created successfully.\")\n\n# ## 8. (Optional) Verify Submission Format\n\n# You might want to load the generated CSV to quickly check its format.\nloaded_submission_df = pd.read_csv('submission.csv')\nprint(f\"Generated submission shape: {loaded_submission_df.shape}\")\nprint(loaded_submission_df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Now, we'll structure the predictions into the required DataFrame format for submission.\n\nsubmission_data = []\nfor oid_ypos, preds in predictions.items():\n    row = {'oid_ypos': oid_ypos}\n    row.update(preds)\n    submission_data.append(row)\n\nsubmission_df = pd.DataFrame(submission_data)\n\n# Ensure the order of columns is correct\nsubmission_cols = ['oid_ypos'] + [f'x_{i}' for i in range(1, 70) if i % 2 != 0]\nsubmission_df = submission_df[submission_cols]","metadata":{}},{"cell_type":"markdown","source":"# Geophysical Waveform Inversion - Submission Notebook\n\nThis notebook provides a basic structure for generating a submission file\nfor the Yale/UNC-CH Geophysical Waveform Inversion competition on Kaggle.\n\n## 1. Import Libraries\n\n```python\nimport pandas as pd\nimport numpy as np\n# Add other necessary libraries here (e.g., your ML framework)\n# Example:\n# from sklearn.ensemble import RandomForestRegressor\nimport joblib![](http://)","metadata":{}}]}