{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":39763,"databundleVersionId":11756775,"sourceType":"competition"},{"sourceId":11672208,"sourceType":"datasetVersion","datasetId":4402985},{"sourceId":248942948,"sourceType":"kernelVersion"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"M1 1  ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport os, gc\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import KFold\nimport sklearn\nimport matplotlib.pyplot as plt\nimport pickle\nimport shutil\n\nimport time\n\nimport scipy.stats as stats\nimport math\nfrom multiprocessing import Pool","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-06T15:09:02.940862Z","iopub.execute_input":"2025-06-06T15:09:02.941254Z","iopub.status.idle":"2025-06-06T15:09:06.705667Z","shell.execute_reply.started":"2025-06-06T15:09:02.94122Z","shell.execute_reply":"2025-06-06T15:09:06.704846Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"env = \"Kaggle\"\nDEBUG = False\nmodel_name = '111'\nTPU = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T15:09:06.706692Z","iopub.execute_input":"2025-06-06T15:09:06.707205Z","iopub.status.idle":"2025-06-06T15:09:06.712769Z","shell.execute_reply.started":"2025-06-06T15:09:06.707175Z","shell.execute_reply":"2025-06-06T15:09:06.711239Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Connect to drive/Save folder","metadata":{}},{"cell_type":"code","source":"import os\nimport json\n\nsave_folder_name = f'models/model {model_name}'\n\nif env == 'Kaggle':\n    base_folder = '/kaggle/working'\n    save_folder = '/kaggle/data'\n    try:\n        os.mkdir(save_folder)\n    except Exception as e:\n        print('exception error:')\n        print(e)\n    print(os.listdir('/kaggle'))\n    f = open('/kaggle/input/kaggle-json/kaggle.json')\n    kaggle_json = json.load(f)\n    KAGGLE_USERNAME = kaggle_json['username']\n    KAGGLE_KEY = kaggle_json['key']\n    os.environ[\"KAGGLE_USERNAME\"] = KAGGLE_USERNAME\n    os.environ[\"KAGGLE_KEY\"] = KAGGLE_KEY\nelif env == 'Colab':\n    from google.colab import drive\n    drive.mount('/content/drive')\n    save_folder = '/content/save_folder'\n    try:\n        os.mkdir(save_folder)\n    except Exception as e:\n        print('exception error:')\n        print(e)\n\n    f = open('/content/drive/MyDrive/kaggle/kaggle_auth/kaggle.json')\n    kaggle_json = json.load(f)\n\n    KAGGLE_USERNAME = kaggle_json['username']\n    KAGGLE_KEY = kaggle_json['key']\n\n    os.environ[\"KAGGLE_USERNAME\"] = KAGGLE_USERNAME\n    os.environ[\"KAGGLE_KEY\"] = KAGGLE_KEY","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T15:09:06.715095Z","iopub.execute_input":"2025-06-06T15:09:06.715357Z","iopub.status.idle":"2025-06-06T15:09:06.750071Z","shell.execute_reply.started":"2025-06-06T15:09:06.715336Z","shell.execute_reply":"2025-06-06T15:09:06.748729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Download data/buckets/TFRecords path","metadata":{}},{"cell_type":"code","source":"folders = [x for x in os.listdir('/kaggle/input/') if x.startswith('gwi')]\nfolders","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T15:09:06.751196Z","iopub.execute_input":"2025-06-06T15:09:06.75154Z","iopub.status.idle":"2025-06-06T15:09:06.762872Z","shell.execute_reply.started":"2025-06-06T15:09:06.751508Z","shell.execute_reply":"2025-06-06T15:09:06.761861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def save_to_tfrecord(file_name, vel, seis, batch_index):\n    import tensorflow as tf\n    with tf.io.TFRecordWriter(file_name, 'GZIP') as file_writer:\n        for index in range(len(seis)):\n            data_cols = seis[index]\n            data_cols_targets = vel[index]\n            \n            data_cols = tf.io.serialize_tensor(data_cols).numpy()\n            data_cols_targets = tf.io.serialize_tensor(data_cols_targets).numpy()\n            \n            features = {}\n            features['x1'] = tf.train.Feature(bytes_list=tf.train.BytesList(value=[data_cols]))\n            features['x2'] = tf.train.Feature(bytes_list=tf.train.BytesList(value=[data_cols_targets]))\n            features['idx'] = tf.train.Feature(int64_list=tf.train.Int64List(value=[index]))\n            features['batch_index'] = tf.train.Feature(int64_list=tf.train.Int64List(value=[1000000+batch_index]))\n            record_bytes = tf.train.Example(features=tf.train.Features(feature=features)).SerializeToString()\n            file_writer.write(record_bytes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T15:09:06.764138Z","iopub.execute_input":"2025-06-06T15:09:06.764526Z","iopub.status.idle":"2025-06-06T15:09:06.782411Z","shell.execute_reply.started":"2025-06-06T15:09:06.764464Z","shell.execute_reply":"2025-06-06T15:09:06.78143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_num = 50\ndef save_file_tfrecords(folder, file_idx, k):\n    import tensorflow as tf\n    vel = pickle.load(open(f'/kaggle/input/{folder}/vel_{file_idx}.p', 'br'))\n    seis = pickle.load(open(f'/kaggle/input/{folder}/seis_{file_idx}.p', 'br'))\n\n    if len(vel)>0:\n        vel = tf.reverse(vel, axis = [2])\n        seis = tf.reverse(seis, axis = [1,3])\n    \n        for i in range(len(vel)//batch_num):\n            batch_vels = vel[i*batch_num:(i+1)*batch_num]\n            batch_seises = seis[i*batch_num:(i+1)*batch_num]\n            i_total = file_idx*len(vel)//batch_num+i\n            tffile_name = f\"{save_folder}/{k}/tfds/{i_total}.tfrecord\"\n            save_to_tfrecord(tffile_name, batch_vels, batch_seises , i_total)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T15:09:06.783547Z","iopub.execute_input":"2025-06-06T15:09:06.783846Z","iopub.status.idle":"2025-06-06T15:09:06.801821Z","shell.execute_reply.started":"2025-06-06T15:09:06.783823Z","shell.execute_reply":"2025-06-06T15:09:06.800919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_file(args):\n    file_idx, folder, k = args\n    print(f'{file_idx}\\n')\n    save_file_tfrecords(folder, file_idx, k)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T15:09:06.802863Z","iopub.execute_input":"2025-06-06T15:09:06.803186Z","iopub.status.idle":"2025-06-06T15:09:06.818051Z","shell.execute_reply.started":"2025-06-06T15:09:06.803164Z","shell.execute_reply":"2025-06-06T15:09:06.81703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nfor k in range(len(folders)):\n    print(f'{k}: k')\n    try:\n        os.mkdir(f\"{save_folder}/{k}\")\n        os.mkdir(f\"{save_folder}/{k}/tfds\")\n    except:\n        pass\n    folder = folders[k]\n    \n    files_in_folders = os.listdir(f'/kaggle/input/{folder}')\n    seis_files = [x for x in files_in_folders if x.startswith('seis')]\n    files_num = len(seis_files)\n\n    with Pool(processes=4) as pool:\n        pool.map(process_file, [(i, folder, k) for i in range(files_num)])\n        \n    shutil.make_archive(f'{save_folder}/{k}', 'zip', f'{save_folder}/{k}')\n    shutil.rmtree(f'{save_folder}/{k}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-06T15:09:06.819166Z","iopub.execute_input":"2025-06-06T15:09:06.81946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle datasets init -p $save_folder\n\nwith open(f'{save_folder}/dataset-metadata.json', 'r+') as f:\n    data = json.load(f)\n    data['title'] = f'GWI: TFRecs Example'\n    data['id'] = f'shlomoron/gwi-tfrecs-Example'\n    data['licenses'][0]['name'] = 'CC0-1.0'\n    f.seek(0)        # <--- should reset file position to the beginning.\n    json.dump(data, f, indent=4)\n    f.truncate()     # remove remaining part\n\n!kaggle datasets create -p $f'{save_folder}' -u --dir-mode zip","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}