{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport bson\nimport itertools\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport gc","metadata":{"execution":{"iopub.status.busy":"2022-05-23T20:42:44.577803Z","iopub.execute_input":"2022-05-23T20:42:44.578643Z","iopub.status.idle":"2022-05-23T20:42:44.751367Z","shell.execute_reply.started":"2022-05-23T20:42:44.578538Z","shell.execute_reply":"2022-05-23T20:42:44.750363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"class Config:\n    DATA_DIR = \"../input/cdiscount-image-classification-challenge\"\n    NPRODS_TRAIN = 7_069_896\n    NPRODS_TEST = 1_768_182\n    GC_LIMIT = 1000\n    DECAY = 100_000\n    DECAY_AMOUNT = 0.99\n    \n    @classmethod\n    def get_data_path(cls, filename):\n        return os.path.join(cls.DATA_DIR, filename)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-23T20:42:44.753309Z","iopub.execute_input":"2022-05-23T20:42:44.753625Z","iopub.status.idle":"2022-05-23T20:42:44.759699Z","shell.execute_reply.started":"2022-05-23T20:42:44.753566Z","shell.execute_reply":"2022-05-23T20:42:44.758784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_metadata(filename, *, nprods=None, has_labels=True):\n    filepath = Config.get_data_path(filename)\n    \n    gc_limit = Config.GC_LIMIT\n    decay = Config.DECAY\n    decay_amt = Config.DECAY_AMOUNT\n    \n    gc_limit_minus_1, decay_minus_1 = gc_limit - 1, decay - 1\n        \n    with open(filepath, 'rb') as f:\n        bdata = bson.decode_file_iter(f)\n        \n        if nprods is not None:\n            # Limit the number of records\n            bdata = itertools.islice(bdata, None, nprods)\n        else:\n            nprods = Config.NPRODS_TRAIN if \"train\" in filename else Config.NPRODS_TEST\n            \n        pbdata = tqdm(bdata, total=nprods, desc=filename)\n        \n        metadata = []\n        \n        # Record the current position\n        curr = f.tell()\n        \n        for idx, d in enumerate(pbdata):\n            # Get the starting position\n            # And change the current position since once bson decodes a record\n            # File pointer has already moved to the next record\n            start, curr = curr, f.tell()\n            \n            # Get length of record\n            length = curr - start\n            \n            record = (d[\"_id\"], start, length, len(d[\"imgs\"]))\n            \n            if has_labels is True:\n                record += (d[\"category_id\"],)\n            \n            metadata.append(record)\n            \n            # To manage RAM usage\n            del d\n            \n            # Force garbage collection\n            if idx % gc_limit == gc_limit_minus_1:\n                gc.collect()\n                \n            # Increase the frequency of garbage collection as time progresses\n            if idx % decay == decay_minus_1:\n                gc_limit *= decay_amt\n                gc_limit = int(gc_limit)\n            \n        return metadata","metadata":{"execution":{"iopub.status.busy":"2022-05-23T20:42:44.760906Z","iopub.execute_input":"2022-05-23T20:42:44.761694Z","iopub.status.idle":"2022-05-23T20:42:44.774058Z","shell.execute_reply.started":"2022-05-23T20:42:44.761643Z","shell.execute_reply":"2022-05-23T20:42:44.772915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_csv(filename, *, nprods=None, has_labels=True):\n    metadata = get_metadata(filename, nprods=nprods, has_labels=has_labels)\n    \n    gc.collect()\n    \n    cols = [\"pid\", \"start\", \"length\", \"n_imgs\"]\n    \n    if has_labels is True:\n        cols.append(\"category_id\")\n    \n    df = pd.DataFrame(metadata, columns=cols)\n    \n    name, _ = os.path.splitext(filename)\n    dest = f\"{name}_metadata.csv\"\n    \n    df.to_csv(dest, index=False)\n    \n    del metadata\n    del df\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T20:42:44.777760Z","iopub.execute_input":"2022-05-23T20:42:44.778169Z","iopub.status.idle":"2022-05-23T20:42:44.788742Z","shell.execute_reply.started":"2022-05-23T20:42:44.778137Z","shell.execute_reply":"2022-05-23T20:42:44.788052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"make_csv(\"train.bson\")\ngc.collect()\nmake_csv(\"test.bson\", has_labels=False)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-05-23T20:43:08.299130Z","iopub.execute_input":"2022-05-23T20:43:08.299656Z","iopub.status.idle":"2022-05-23T20:43:30.514888Z","shell.execute_reply.started":"2022-05-23T20:43:08.299620Z","shell.execute_reply":"2022-05-23T20:43:30.513220Z"},"trusted":true},"execution_count":null,"outputs":[]}]}