{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Note - *This is my first kaggle notebook and also first sompetitions, if anyone has any king of suggestions for improvement please share it with me to get better in this.*","metadata":{}},{"cell_type":"markdown","source":"# **Imports**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport glob","metadata":{"execution":{"iopub.status.busy":"2023-09-23T18:48:33.085654Z","iopub.execute_input":"2023-09-23T18:48:33.085995Z","iopub.status.idle":"2023-09-23T18:48:33.091865Z","shell.execute_reply.started":"2023-09-23T18:48:33.085967Z","shell.execute_reply":"2023-09-23T18:48:33.090482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **File's Structure**\n**(In Tree Format)**","metadata":{}},{"cell_type":"code","source":"!tree -I *.npz /kaggle/input/predict-ai-model-runtime/npz_all","metadata":{"execution":{"iopub.status.busy":"2023-09-23T18:48:33.103338Z","iopub.execute_input":"2023-09-23T18:48:33.103968Z","iopub.status.idle":"2023-09-23T18:48:37.340593Z","shell.execute_reply.started":"2023-09-23T18:48:33.103902Z","shell.execute_reply":"2023-09-23T18:48:37.339429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are .npz files which are divided in two parts, that is - layout and tile.\n\nIn layout, there is further parts which is nlp and xla. Both consist default and random.","metadata":{}},{"cell_type":"markdown","source":"# **Files**","metadata":{}},{"cell_type":"code","source":"all_files = glob.glob(\"/kaggle/input/predict-ai-model-runtime/npz_all/npz/**/*.npz\", recursive=True)\ntrain_files = glob.glob('/kaggle/input/predict-ai-model-runtime/npz_all/npz/**/train/*', recursive=True)\nval_files = glob.glob('/kaggle/input/predict-ai-model-runtime/npz_all/npz/**/valid/*', recursive=True)\ntest_files = glob.glob('/kaggle/input/predict-ai-model-runtime/npz_all/npz/**/test/*', recursive=True)\n\nprint(\"Total Files are \", [len(all_files)], \"out of which\")\nprint(\"Training Files are \", [len(train_files)])\nprint(\"Testing Files are \", [len(test_files)])\nprint(\"Validation Files are \", [len(val_files)])","metadata":{"execution":{"iopub.status.busy":"2023-09-23T18:48:37.342786Z","iopub.execute_input":"2023-09-23T18:48:37.343746Z","iopub.status.idle":"2023-09-23T18:48:37.495847Z","shell.execute_reply.started":"2023-09-23T18:48:37.343709Z","shell.execute_reply":"2023-09-23T18:48:37.494350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This are the number of files of each split.\n\nWe'll use this splits in our training model in the future.","metadata":{}},{"cell_type":"markdown","source":"# **Sample Files**","metadata":{}},{"cell_type":"code","source":"print(\"Sample File of TILE\")\nfor ex in train_files:\n    if \"tile\" in ex:\n        data = np.load(ex)\n        for key in data.files:\n            print(key.ljust(15), '\\t:', data[key].shape)\n        break\n        \nprint(\"\\n\\nSample File of LAYOUT\")\nfor ex in train_files:\n    if \"layout\" in ex:\n        data = np.load(ex)\n        for key in data.files:\n            print(key.ljust(15), '\\t:', data[key].shape)\n        break","metadata":{"execution":{"iopub.status.busy":"2023-09-23T18:48:37.497596Z","iopub.execute_input":"2023-09-23T18:48:37.497991Z","iopub.status.idle":"2023-09-23T18:48:37.902516Z","shell.execute_reply.started":"2023-09-23T18:48:37.497962Z","shell.execute_reply":"2023-09-23T18:48:37.901018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Below code will give the sample output of the .npz file of tile in the form of dictionary.**","metadata":{}},{"cell_type":"code","source":"tile_data = np.load(\"/kaggle/input/predict-ai-model-runtime/npz_all/npz/tile/xla/train/alexnet_train_batch_32_-1bae27a41d70f4dc.npz\")\ndf = {}\n\nfor feat in tile_data.files:\n    df[feat] = tile_data[feat]\n\nfor key in df:\n    print('\\n', key, \": \\n\", df[key], '\\n', \"shape - \", (df[key].shape))\n\nprint(type(df))","metadata":{"execution":{"iopub.status.busy":"2023-09-23T18:51:11.252735Z","iopub.execute_input":"2023-09-23T18:51:11.253275Z","iopub.status.idle":"2023-09-23T18:51:11.270262Z","shell.execute_reply.started":"2023-09-23T18:51:11.253253Z","shell.execute_reply":"2023-09-23T18:51:11.268764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Below code will give the sample output of the .npz file of layout nlp in the form of dictionary.**","metadata":{}},{"cell_type":"code","source":"layout_nlp_data = np.load(\"/kaggle/input/predict-ai-model-runtime/npz_all/npz/layout/nlp/default/train/albert_en_base_batch_size_16_test.npz\")\ndf = {}\n\nfor feat in layout_nlp_data.files:\n    df[feat] = layout_nlp_data[feat]\n\nfor key in df:\n    print('\\n', key, \": \\n\", df[key], '\\n', \"shape - \", (df[key].shape))\n\nprint(type(df))","metadata":{"execution":{"iopub.status.busy":"2023-09-23T18:51:34.554914Z","iopub.execute_input":"2023-09-23T18:51:34.555279Z","iopub.status.idle":"2023-09-23T18:51:35.483779Z","shell.execute_reply.started":"2023-09-23T18:51:34.555252Z","shell.execute_reply":"2023-09-23T18:51:35.483086Z"},"trusted":true},"execution_count":null,"outputs":[]}]}