{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# You can use this notebook to start working with only an small subset of the training data provided.\n\n# The code takes care of loading only the subset of rows from the train_meta file which are relevant\n# to the batches that you are going to work with, avoiding wasting valuable ram memory\n\n# Basically, you only have to choose the number of train batches you are going to use\n\n# This is useful for making small experiments with some of the data and bootstrapping your project","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:28:39.541495Z","iopub.execute_input":"2023-03-08T23:28:39.542304Z","iopub.status.idle":"2023-03-08T23:28:39.568811Z","shell.execute_reply.started":"2023-03-08T23:28:39.542178Z","shell.execute_reply":"2023-03-08T23:28:39.567683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyarrow.parquet as pq\n\nimport random\nimport glob\nimport os\nimport re\n\nimport pandas as pd\nimport numpy as np\n\nimport tensorflow as tf\nimport keras","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:28:39.570762Z","iopub.execute_input":"2023-03-08T23:28:39.571148Z","iopub.status.idle":"2023-03-08T23:28:46.499170Z","shell.execute_reply.started":"2023-03-08T23:28:39.571116Z","shell.execute_reply":"2023-03-08T23:28:46.497856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##############################################################################\n#\n# Functions\n#\n##############################################################################\n\ndef batch_id_from_batch_path(x):\n    # Takes the path to a batch file\n    # Gives you back the number identifying it\n    filename = os.path.basename(x)\n    m = re.match('batch_(\\d+).parquet', filename)\n    if m:\n        return m.group(1)\n    else:\n        return None\n\n\ndef batch_path_from_batch_id(x):\n    # Takes a number and produces the name of the corresponding batch file\n    return f'batch_{x}.parquet'","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:28:46.500425Z","iopub.execute_input":"2023-03-08T23:28:46.501038Z","iopub.status.idle":"2023-03-08T23:28:46.508967Z","shell.execute_reply.started":"2023-03-08T23:28:46.501003Z","shell.execute_reply":"2023-03-08T23:28:46.506679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##############################################################################\n#\n# Parameters\n#\n##############################################################################\n\n# Folder containing the batch files\nfolder_train_batches = r'/kaggle/input/icecube-neutrinos-in-deep-ice/train'\n\n# Mask that matches the format of the name of the batch files\nmask_train_batches = 'batch*.parquet'\n\n# The path of the train_meta file\ntrain_meta_path = r'/kaggle/input/icecube-neutrinos-in-deep-ice/train_meta.parquet'\n\n# How many train batches do you want to load. If None or negative, everyone will be loaded\nnumber_of_train_batches = 1\n\n# Choose the first files in the folder or randomize them\nrandomize_batches_order = True\n\n# Whether or no to use more than one thread to load the train_meta file\nuse_threads = True\n\n# The column by which the meta file is going to be filtered\ncolumn_filter = 'batch_id'","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:28:46.511782Z","iopub.execute_input":"2023-03-08T23:28:46.512517Z","iopub.status.idle":"2023-03-08T23:28:46.523289Z","shell.execute_reply.started":"2023-03-08T23:28:46.512476Z","shell.execute_reply":"2023-03-08T23:28:46.522237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We get the list of all the files in the folder matching the provided mask\nlist_all_train_batches = glob.glob(os.path.join(folder_train_batches, mask_train_batches))\n\n# Randomize the order of the files in that folder\nif randomize_batches_order:\n    random.shuffle(list_all_train_batches)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:28:46.524584Z","iopub.execute_input":"2023-03-08T23:28:46.524906Z","iopub.status.idle":"2023-03-08T23:28:46.639968Z","shell.execute_reply.started":"2023-03-08T23:28:46.524877Z","shell.execute_reply":"2023-03-08T23:28:46.638720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check whether we are to filter or not the files\n# We also check that we are not filtering more files than there are\nif number_of_train_batches is None or number_of_train_batches < 0 or number_of_train_batches >= len(list_all_train_batches):\n    filter_train_data = False\n    list_train_batches = list_all_train_batches\nelse:\n    filter_train_data = True\n    list_train_batches = list_all_train_batches\n    list_train_batches = list_all_train_batches[0:number_of_train_batches]\n\n    # Build the filter object which will be used to pick\n    # the rows to load from the meta file\n    values_filter = [batch_id_from_batch_path(x) for x in list_train_batches]\n    filter_for_meta = [column_filter, 'in', values_filter]\n\n    # We use only one filter to select the rows of the train_meta file\n    list_of_filters = [filter_for_meta]","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:28:46.642680Z","iopub.execute_input":"2023-03-08T23:28:46.643144Z","iopub.status.idle":"2023-03-08T23:28:46.650747Z","shell.execute_reply.started":"2023-03-08T23:28:46.643099Z","shell.execute_reply":"2023-03-08T23:28:46.649569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the meta file, choosing or not specific rows according to the filter\nif filter_train_data:\n    train_meta_df = pq.read_table(train_meta_path, use_threads=use_threads, filters=list_of_filters).to_pandas()\nelse:\n    train_meta_df = pq.read_table(train_meta_path, use_threads=use_threads,).to_pandas()\n    \ntrain_meta_df = train_meta_df.set_index('event_id')","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:28:46.652498Z","iopub.execute_input":"2023-03-08T23:28:46.652863Z","iopub.status.idle":"2023-03-08T23:28:59.625315Z","shell.execute_reply.started":"2023-03-08T23:28:46.652828Z","shell.execute_reply":"2023-03-08T23:28:59.624283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You now have a list of train batches to load\n# You can use it as an entry to a Tensorflow generator\n# Or as the set of a loop to process them one by one\n\nlist_train_batches","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:28:59.626896Z","iopub.execute_input":"2023-03-08T23:28:59.627577Z","iopub.status.idle":"2023-03-08T23:28:59.636664Z","shell.execute_reply.started":"2023-03-08T23:28:59.627537Z","shell.execute_reply":"2023-03-08T23:28:59.635446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You also have a dataframe containing the meta information\n# only for the events contained in the chosen train_batches\n\ntrain_meta_df","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:28:59.638548Z","iopub.execute_input":"2023-03-08T23:28:59.639463Z","iopub.status.idle":"2023-03-08T23:28:59.666849Z","shell.execute_reply.started":"2023-03-08T23:28:59.639415Z","shell.execute_reply":"2023-03-08T23:28:59.665535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for path_load in list_train_batches:\n    try:\n        load = pd.concat([load, pq.read_table(path_load, use_threads=use_threads).to_pandas()], axis = 0)\n    except:\n        load = pq.read_table(path_load, use_threads=use_threads).to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:28:59.669710Z","iopub.execute_input":"2023-03-08T23:28:59.670497Z","iopub.status.idle":"2023-03-08T23:29:01.551664Z","shell.execute_reply.started":"2023-03-08T23:28:59.670459Z","shell.execute_reply":"2023-03-08T23:29:01.550655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load['auxiliary'] = load['auxiliary'] * 1.0","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:29:01.552629Z","iopub.execute_input":"2023-03-08T23:29:01.552948Z","iopub.status.idle":"2023-03-08T23:29:01.780427Z","shell.execute_reply.started":"2023-03-08T23:29:01.552920Z","shell.execute_reply":"2023-03-08T23:29:01.779194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datos = load.groupby('event_id').apply(lambda x:pd.DataFrame.to_numpy(x))","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:29:01.781910Z","iopub.execute_input":"2023-03-08T23:29:01.782396Z","iopub.status.idle":"2023-03-08T23:29:19.677928Z","shell.execute_reply.started":"2023-03-08T23:29:01.782330Z","shell.execute_reply":"2023-03-08T23:29:19.676893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"azimuth = train_meta_df.loc[datos.index]['azimuth']","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:29:19.679487Z","iopub.execute_input":"2023-03-08T23:29:19.680152Z","iopub.status.idle":"2023-03-08T23:29:19.805887Z","shell.execute_reply.started":"2023-03-08T23:29:19.680116Z","shell.execute_reply":"2023-03-08T23:29:19.804617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"maximum = 0\n\nfor r in datos:\n    maximum = max(maximum, len(r))\n    if maximum == len(r):\n        maximum_row = r\n    \npadding_size = maximum","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:29:19.807288Z","iopub.execute_input":"2023-03-08T23:29:19.807687Z","iopub.status.idle":"2023-03-08T23:29:19.983179Z","shell.execute_reply.started":"2023-03-08T23:29:19.807654Z","shell.execute_reply":"2023-03-08T23:29:19.981593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datos = datos[0:10000]\npaddeado = tf.keras.preprocessing.sequence.pad_sequences(datos, padding='pre')","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:29:19.984982Z","iopub.execute_input":"2023-03-08T23:29:19.985409Z","iopub.status.idle":"2023-03-08T23:29:24.694960Z","shell.execute_reply.started":"2023-03-08T23:29:19.985368Z","shell.execute_reply":"2023-03-08T23:29:24.693798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paddeado","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:29:24.696683Z","iopub.execute_input":"2023-03-08T23:29:24.697243Z","iopub.status.idle":"2023-03-08T23:29:24.709067Z","shell.execute_reply.started":"2023-03-08T23:29:24.697209Z","shell.execute_reply":"2023-03-08T23:29:24.707674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_paddeado = paddeado[0:8000]\ntrain_azimuth = azimuth[0:8000]\ntest_paddeado = paddeado[8000:-1]\ntest_azimuth = azimuth[8000:-1]","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:29:24.711210Z","iopub.execute_input":"2023-03-08T23:29:24.712104Z","iopub.status.idle":"2023-03-08T23:29:24.730780Z","shell.execute_reply.started":"2023-03-08T23:29:24.712049Z","shell.execute_reply":"2023-03-08T23:29:24.729622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, SimpleRNN\n\n# Define the RNN model\nmodel = Sequential([\n    SimpleRNN(64, input_shape=(padding_size, 4), activation='relu'),\n    Dense(1, activation='linear')\n])\n\n# Compile the model\nmodel.compile(optimizer='adam', loss='mse')\n\n\n# Train the model\nhistory = model.fit(train_paddeado, train_azimuth, validation_data=(test_paddeado, test_azimuth), epochs=50, batch_size=32)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-08T23:29:24.732202Z","iopub.execute_input":"2023-03-08T23:29:24.732983Z"},"trusted":true},"execution_count":null,"outputs":[]}]}