{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## ISLR: Create TFRecord\n\nIn this notebook, I will create time-series TFRecord for [Google - Isolated Sign Language Recognition Competition](https://www.kaggle.com/competitions/asl-signs). I will choose 12 as sequence_length. The dataset has dynamic sequence length, I will convert to static sequence length by simply using `tf.image.resize` API. [Here](https://www.kaggle.com/code/lonnieqin/isolated-sign-language-recognition-with-convlstm1d/notebook?scriptVersionId=120834686) is a baseline to train with this dataset. ","metadata":{}},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"class CFG:\n    data_path = \"../input/asl-signs/\"\n    rows_per_frame = 543 \n    use_joblib = False","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:44:33.467865Z","iopub.execute_input":"2023-03-28T04:44:33.468244Z","iopub.status.idle":"2023-03-28T04:44:33.497341Z","shell.execute_reply.started":"2023-03-28T04:44:33.468211Z","shell.execute_reply":"2023-03-28T04:44:33.495713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom sklearn.model_selection import StratifiedKFold\nfrom tqdm.notebook import tqdm\nimport json\nimport shutil\nimport os\nfrom pathlib import Path\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:44:33.499290Z","iopub.execute_input":"2023-03-28T04:44:33.499852Z","iopub.status.idle":"2023-03-28T04:44:43.287262Z","shell.execute_reply.started":"2023-03-28T04:44:33.499816Z","shell.execute_reply":"2023-03-28T04:44:43.285746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utilities","metadata":{}},{"cell_type":"code","source":"ROWS_PER_FRAME = 543  # number of landmarks per frame\n\ndef load_relevant_data_subset_with_imputation(pq_path, replace_nan):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    \n    if replace_nan:\n        data.replace(np.nan, 0, inplace=True)\n        \n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    print(data.shape)\n    return data.astype(np.float32)\n\ndef read_dict(file_path):\n    path = os.path.expanduser(file_path)\n    with open(path, \"r\") as f:\n        dic = json.load(f)\n    return dic","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:44:43.288695Z","iopub.execute_input":"2023-03-28T04:44:43.289331Z","iopub.status.idle":"2023-03-28T04:44:43.299622Z","shell.execute_reply.started":"2023-03-28T04:44:43.289298Z","shell.execute_reply":"2023-03-28T04:44:43.297742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(f\"{CFG.data_path}train.csv\")\nlabel_index = read_dict(f\"{CFG.data_path}sign_to_prediction_index_map.json\")\nindex_label = dict([(label_index[key], key) for key in label_index])\nprint(label_index)\ntrain[\"label\"] = train[\"sign\"].map(lambda sign: label_index[sign])\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:44:43.301153Z","iopub.execute_input":"2023-03-28T04:44:43.301446Z","iopub.status.idle":"2023-03-28T04:44:43.566003Z","shell.execute_reply.started":"2023-03-28T04:44:43.301419Z","shell.execute_reply":"2023-03-28T04:44:43.564804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create TF-Record","metadata":{}},{"cell_type":"code","source":"### Create Kaggle Dataset if not exists \nDATASET_NAME = 'asl-tensorflow-record-dataset'\n\n!rm -rf /tmp/{DATASET_NAME}\n\nshutil.rmtree(f'/tmp/{DATASET_NAME}', ignore_errors=True)\nos.makedirs(f'/tmp/{DATASET_NAME}', exist_ok=True)\n\nwith open('../input/my-secret/kaggle.json') as f:\n    kaggle_creds = json.load(f)\n    \nos.environ['KAGGLE_USERNAME'] = kaggle_creds['username']\nos.environ['KAGGLE_KEY'] = kaggle_creds['key']\n\n!kaggle datasets init -p /tmp/{DATASET_NAME}\n\nwith open(f'/tmp/{DATASET_NAME}/dataset-metadata.json') as f:\n    dataset_meta = json.load(f)\n\ndataset_meta['id'] = f'meowmeowmeowmeowmeow/{DATASET_NAME}'\ndataset_meta['title'] = DATASET_NAME\ndataset_meta['note'] = 'Individual tfrecord per sample'\n\nwith open(f'/tmp/{DATASET_NAME}/dataset-metadata.json', \"w\") as outfile:\n    json.dump(dataset_meta, outfile)\n#print(dataset_meta)\n\n!cp /tmp/{DATASET_NAME}/dataset-metadata.json /tmp/{DATASET_NAME}/meta.json\n!ls /tmp/{DATASET_NAME}\n\n#!kaggle datasets create -u -p /tmp/{DATASET_NAME} \n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:44:43.570076Z","iopub.execute_input":"2023-03-28T04:44:43.570408Z","iopub.status.idle":"2023-03-28T04:44:45.094495Z","shell.execute_reply.started":"2023-03-28T04:44:43.570375Z","shell.execute_reply":"2023-03-28T04:44:45.093284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_record(feature, num_frames, label, seq_id):\n    dic = {}\n#     dic[\"feature\"] = tf.train.Feature(float_list=tf.train.FloatList(value=feature))\n    \n    dic[\"feature\"] = tf.train.Feature(bytes_list=tf.train.BytesList(value=[feature]))\n    dic[\"num_frames\"] = tf.train.Feature(int64_list=tf.train.Int64List(value=[num_frames]))\n    dic[\"label\"] = tf.train.Feature(int64_list=tf.train.Int64List(value=[label]))\n    dic[\"id\"] = tf.train.Feature(int64_list=tf.train.Int64List(value=[seq_id]))\n    record_bytes = tf.train.Example(features=tf.train.Features(feature=dic)).SerializeToString()\n    return record_bytes","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:44:45.095895Z","iopub.execute_input":"2023-03-28T04:44:45.096191Z","iopub.status.idle":"2023-03-28T04:44:45.103465Z","shell.execute_reply.started":"2023-03-28T04:44:45.096161Z","shell.execute_reply":"2023-03-28T04:44:45.102342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = train.sequence_id\nlabels = train.label\n\nkfold = StratifiedKFold(n_splits=5)\n\nfolds = list(kfold.split(ids, labels))\n\nfor x, y in folds:\n    overlap = set(x).intersection(set(y))\n    assert not overlap\n    \nfor i, (x1, y1) in enumerate(folds):\n    for j, (x2, y2) in enumerate(folds):\n        overlap = set(y1).intersection(set(y2))\n        if i != j:\n            assert not overlap, 'test set overlaps across folds'\n        \nfor i, (x1, y1) in enumerate(folds):\n    train_set = set()\n    for j, (x2, y2) in enumerate(folds):\n        if i != j:\n            train_set.update(set(y2))\n    assert not set(x1).symmetric_difference(set(train_set))\n            \nfig, ax = plt.subplots(ncols=5, figsize=(20, 4))\nfor i, (x, y) in enumerate(folds):\n    fold_labels = train.iloc[y].label\n    ax[i].hist(fold_labels, label=f'Fold {i}', alpha=0.5)\n    ax[i].legend()\n    \nfolds","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:44:45.105908Z","iopub.execute_input":"2023-03-28T04:44:45.106416Z","iopub.status.idle":"2023-03-28T04:44:46.376026Z","shell.execute_reply.started":"2023-03-28T04:44:45.106367Z","shell.execute_reply":"2023-03-28T04:44:46.374437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_tf_records(sequence_indexes, progress=False, limit=None):\n    iterator = sequence_indexes if not progress else tqdm(sequence_indexes)\n    \n    assert len(sequence_indexes) == 1\n    \n#     df = train[train.sequence_id == seq_idx]\n\n    seq_id = int(train.iloc[sequence_indexes[0]].sequence_id)\n    \n    save_path = f\"/tmp/{DATASET_NAME}/{seq_id}.tfrecords\"\n    with tf.io.TFRecordWriter(save_path) as file_writer:\n        sub_iterator = sequence_indexes if not progress else tqdm(sequence_indexes)\n        for idx, i in enumerate(sub_iterator):\n            path = f\"{CFG.data_path}{train.iloc[i].path}\"\n            feature = load_relevant_data_subset_with_imputation(path, False)\n            feature_data = bytes(feature.data)\n            num_frames = feature.shape[0]\n\n#             print(feature.shape, feature.dtype, type(feature))\n            label = int(train.iloc[i].label)\n            \n            file_writer.write(create_record(feature_data, num_frames, label, seq_id))\n\n#             if limit and idx > limit:\n#                 print(f'BREAK FOLD [{fold_idx}] DUE TO [{limit}] LIIMIT PROVIDED')\n#                 break\n#             if not progress and idx % 50 == 0:\n#                 print(f'FOLD #{fold_idx}: Progress {idx/len(sequence_indexes)*100:0.3f}%')\n\n# create_tf_records(folds[2][1], 0, True, 100)\nfor idx, i in enumerate( train.index):\n    create_tf_records([i])\n    if idx > 5:\n        break\n\nprint(f'Size of the dataset: {sum(os.path.getsize(x) for x in Path(\"/tmp/asl-tensorflow-record-dataset\").glob(\"*.tfrecords\")) /1024/1024/1024:0.2f} GB')","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:44:46.378169Z","iopub.execute_input":"2023-03-28T04:44:46.379488Z","iopub.status.idle":"2023-03-28T04:44:46.610607Z","shell.execute_reply.started":"2023-03-28T04:44:46.379429Z","shell.execute_reply":"2023-03-28T04:44:46.609496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !rm -rf /tmp/asl-tensorflow-record-dataset/*","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:44:46.611778Z","iopub.execute_input":"2023-03-28T04:44:46.612147Z","iopub.status.idle":"2023-03-28T04:44:46.618226Z","shell.execute_reply.started":"2023-03-28T04:44:46.612115Z","shell.execute_reply":"2023-03-28T04:44:46.616546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folds_seq_idxs = []\nfolds_idxs = []\n\nfor idx, (x, y) in enumerate(folds):\n    folds_seq_idxs.append(y)\n    folds_idxs.append(idx)\n    print(f'Fold #{idx}: {y}')\n    \n# if CFG.use_joblib:\n#     print('Use joblib')\n#     import joblib\n#     _ = joblib.Parallel(n_jobs=3)(\n#             joblib.delayed(create_tf_records)(seq_idxs, fold_idx) for seq_idxs, fold_idx in tqdm(zip(folds_seq_idxs, folds_idxs), total=5)\n#         )\nelse:\n    print('Use single thread')\n#     for seq_idxs, fold_idx in tqdm(zip(folds_seq_idxs, folds_idxs), total=5):\n    for idx, i in tqdm( enumerate( train.index), total=len(train)):\n        create_tf_records([i])","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:44:46.619734Z","iopub.execute_input":"2023-03-28T04:44:46.620164Z","iopub.status.idle":"2023-03-28T04:45:37.678455Z","shell.execute_reply.started":"2023-03-28T04:44:46.620127Z","shell.execute_reply":"2023-03-28T04:45:37.676796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !ls -al /tmp/asl-tensorflow-record-dataset/ | grep tfrecords\n!ls -al /tmp/asl-tensorflow-record-dataset/ | grep tfrecords | wc -l\nprint(f'Size of the dataset: {sum(os.path.getsize(x) for x in Path(\"/tmp/asl-tensorflow-record-dataset\").glob(\"*.tfrecords\")) /1024/1024/1024:0.2f} GB')","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:45:40.407002Z","iopub.execute_input":"2023-03-28T04:45:40.407667Z","iopub.status.idle":"2023-03-28T04:45:40.743703Z","shell.execute_reply.started":"2023-03-28T04:45:40.407618Z","shell.execute_reply":"2023-03-28T04:45:40.741527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_function(record_bytes):\n    return tf.io.parse_single_example(\n          # Data\n          record_bytes,\n          # Schema\n          {\n              \"feature\": tf.io.FixedLenFeature([], dtype=tf.string),\n              \"num_frames\": tf.io.FixedLenFeature([], dtype=tf.int64),\n              \"label\": tf.io.FixedLenFeature([], dtype=tf.int64),\n              \"id\": tf.io.FixedLenFeature([], dtype=tf.int64)\n          }\n      )\ndef preprocess(item):\n    features = item[\"feature\"]\n    features = tf.io.decode_raw(features, tf.float32)\n    features = tf.reshape(features, [item['num_frames'], 543, 3])\n#     features = tf.reshape(features, (CFG.sequence_length, 543, 3))\n    return features, item[\"label\"], item[\"id\"]\n\ndef make_dataset(file_paths, batch_size=128, mode=\"train\"):\n    ds = tf.data.TFRecordDataset(file_paths)\n    ds = ds.map(decode_function)\n    ds = ds.map(preprocess)\n    \n    return ds\n\ntf_records = list(Path(f'/tmp/{DATASET_NAME}/').glob('1000035562.tfrecords'))\ndf = make_dataset(tf_records)\nnext(iter(df.skip(0)))","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:45:43.255009Z","iopub.execute_input":"2023-03-28T04:45:43.255410Z","iopub.status.idle":"2023-03-28T04:45:43.680224Z","shell.execute_reply.started":"2023-03-28T04:45:43.255371Z","shell.execute_reply":"2023-03-28T04:45:43.679116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:45:46.226235Z","iopub.execute_input":"2023-03-28T04:45:46.226651Z","iopub.status.idle":"2023-03-28T04:46:22.092415Z","shell.execute_reply.started":"2023-03-28T04:45:46.226616Z","shell.execute_reply":"2023-03-28T04:46:22.090391Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /tmp/ziped-dataset\n!zip -r /tmp/ziped-dataset/dataset.zip /tmp/{DATASET_NAME}/\n!cp /tmp/{DATASET_NAME}/dataset-metadata.json /tmp/ziped-dataset\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:51:30.334213Z","iopub.execute_input":"2023-03-28T04:51:30.334640Z","iopub.status.idle":"2023-03-28T04:52:09.955844Z","shell.execute_reply.started":"2023-03-28T04:51:30.334605Z","shell.execute_reply":"2023-03-28T04:52:09.953906Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /tmp/ziped-dataset/","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:51:21.696110Z","iopub.execute_input":"2023-03-28T04:51:21.696524Z","iopub.status.idle":"2023-03-28T04:51:21.978315Z","shell.execute_reply.started":"2023-03-28T04:51:21.696483Z","shell.execute_reply":"2023-03-28T04:51:21.976372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\nversion_name = datetime.now().strftime(\"%Y%m%d-%H%M%S\")\n\nfrom time import time\n\ntimestamp = time()\ndate_time = datetime.fromtimestamp(timestamp)\ntimestamp_formatted = date_time.strftime(\"%d_%B_%Y-%H:%M:%S\")\nversion_name = timestamp_formatted\n\nprint(version_name)\n\n!kaggle datasets version -m {version_name} -p /tmp/ziped-dataset","metadata":{"execution":{"iopub.status.busy":"2023-03-28T04:52:31.976031Z","iopub.execute_input":"2023-03-28T04:52:31.977719Z","iopub.status.idle":"2023-03-28T04:52:36.573362Z","shell.execute_reply.started":"2023-03-28T04:52:31.977651Z","shell.execute_reply":"2023-03-28T04:52:36.571360Z"},"trusted":true},"execution_count":null,"outputs":[]}]}