{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"print(\"Please upvote :)\")","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.027036,"end_time":"2022-04-03T08:10:27.867611","exception":false,"start_time":"2022-04-03T08:10:27.840575","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-07T06:53:41.783668Z","iopub.execute_input":"2022-04-07T06:53:41.78424Z","iopub.status.idle":"2022-04-07T06:53:41.805372Z","shell.execute_reply.started":"2022-04-07T06:53:41.784148Z","shell.execute_reply":"2022-04-07T06:53:41.804709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport tensorflow as tf\nfrom tqdm import tqdm\nimport numpy as np\nfrom sklearn.utils import shuffle\nimport os","metadata":{"papermill":{"duration":6.217337,"end_time":"2022-04-03T08:10:34.099209","exception":false,"start_time":"2022-04-03T08:10:27.881872","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-07T06:53:41.806749Z","iopub.execute_input":"2022-04-07T06:53:41.807208Z","iopub.status.idle":"2022-04-07T06:53:47.631891Z","shell.execute_reply.started":"2022-04-07T06:53:41.807176Z","shell.execute_reply":"2022-04-07T06:53:47.630725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"directory = \"../input/happywhale-resized-images/happywhale_resized_images\"\nfilenames = os.listdir(directory)\nfilenames = shuffle(filenames, random_state=0)\nfilenames[:5], len(filenames)","metadata":{"papermill":{"duration":0.932242,"end_time":"2022-04-03T08:10:35.041751","exception":false,"start_time":"2022-04-03T08:10:34.109509","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-07T06:53:47.633648Z","iopub.execute_input":"2022-04-07T06:53:47.634088Z","iopub.status.idle":"2022-04-07T06:53:48.472567Z","shell.execute_reply.started":"2022-04-07T06:53:47.634028Z","shell.execute_reply":"2022-04-07T06:53:48.471897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepath = \"../input/happy-whale-and-dolphin/train.csv\"\ntrain = pd.read_csv(filepath)\ntrain","metadata":{"papermill":{"duration":0.149727,"end_time":"2022-04-03T08:10:35.201639","exception":false,"start_time":"2022-04-03T08:10:35.051912","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-07T06:53:48.473465Z","iopub.execute_input":"2022-04-07T06:53:48.474129Z","iopub.status.idle":"2022-04-07T06:53:48.57778Z","shell.execute_reply.started":"2022-04-07T06:53:48.474098Z","shell.execute_reply":"2022-04-07T06:53:48.577227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_unique_species = sorted(train.species.unique())\ntrain.species = train.species.apply(lambda x: sorted_unique_species.index(x))\ntrain","metadata":{"papermill":{"duration":0.081473,"end_time":"2022-04-03T08:10:35.294095","exception":false,"start_time":"2022-04-03T08:10:35.212622","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-07T06:53:48.579169Z","iopub.execute_input":"2022-04-07T06:53:48.579496Z","iopub.status.idle":"2022-04-07T06:53:48.63317Z","shell.execute_reply.started":"2022-04-07T06:53:48.579461Z","shell.execute_reply":"2022-04-07T06:53:48.632126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bytes_feature(value):\n    if isinstance(value, type(tf.constant(0))):\n        value = value.numpy()\n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=[value]))","metadata":{"papermill":{"duration":0.020735,"end_time":"2022-04-03T08:10:35.326841","exception":false,"start_time":"2022-04-03T08:10:35.306106","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-07T06:53:48.634486Z","iopub.execute_input":"2022-04-07T06:53:48.63498Z","iopub.status.idle":"2022-04-07T06:53:48.64051Z","shell.execute_reply.started":"2022-04-07T06:53:48.634936Z","shell.execute_reply":"2022-04-07T06:53:48.639552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split = np.linspace(0, len(filenames), 23, dtype=int)\n\nfor i in tqdm(range(len(split) - 1)):\n    with tf.io.TFRecordWriter(f\"classification_{i}.tfrecords\") as writer:\n        for j in range(split[i], split[i + 1]):\n            filename = filenames[i]\n            filepath = os.path.join(directory, filename)\n            image_bytes = tf.io.read_file(filepath)\n            image = tf.io.decode_jpeg(image_bytes)\n            image_bytes = bytes(image)\n            species = train.species[train.image == filename].values[0]\n            species = np.array([species], \"int32\")\n            species_bytes = bytes(species)\n            feature = {\n                \"input\": bytes_feature(image_bytes),\n                \"output\": bytes_feature(species_bytes),\n            }\n            tf_example = tf.train.Example(features=tf.train.Features(feature=feature))\n            writer.write(tf_example.SerializeToString())","metadata":{"papermill":{"duration":452.570178,"end_time":"2022-04-03T08:18:07.909","exception":false,"start_time":"2022-04-03T08:10:35.338822","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-04-07T06:53:48.6415Z","iopub.execute_input":"2022-04-07T06:53:48.642053Z"},"trusted":true},"execution_count":null,"outputs":[]}]}