{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nfolder = '.'\nfilepaths = [os.path.join(folder, f) for f in os.listdir(folder)]","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:45.160366Z","iopub.execute_input":"2022-12-29T12:48:45.160732Z","iopub.status.idle":"2022-12-29T12:48:45.166797Z","shell.execute_reply.started":"2022-12-29T12:48:45.1607Z","shell.execute_reply":"2022-12-29T12:48:45.165603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filepaths = [f.path for f in os.scandir('.') if f.is_file()]\ndirpaths  = [f.path for f in os.scandir('.') if f.is_dir()]","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:45.173586Z","iopub.execute_input":"2022-12-29T12:48:45.173864Z","iopub.status.idle":"2022-12-29T12:48:45.179364Z","shell.execute_reply.started":"2022-12-29T12:48:45.173837Z","shell.execute_reply":"2022-12-29T12:48:45.178387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for (dirpath, dirnames, filenames) in os.walk('.'):\n    for f in filenames:\n        print('FILE :', os.path.join(dirpath, f))\n    for d in dirnames:\n        print('DIRECTORY :', os.path.join(dirpath, d))","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:45.184127Z","iopub.execute_input":"2022-12-29T12:48:45.184393Z","iopub.status.idle":"2022-12-29T12:48:45.192228Z","shell.execute_reply.started":"2022-12-29T12:48:45.184369Z","shell.execute_reply":"2022-12-29T12:48:45.191157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\nfilepaths = glob.glob('./assets/*.mp4')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:45.196535Z","iopub.execute_input":"2022-12-29T12:48:45.196823Z","iopub.status.idle":"2022-12-29T12:48:45.201259Z","shell.execute_reply.started":"2022-12-29T12:48:45.196797Z","shell.execute_reply":"2022-12-29T12:48:45.200188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pathlib \npath = pathlib.Path.home() / 'src'\nfor p in path.glob('*.py'):\n    print(p.name)","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:45.219961Z","iopub.execute_input":"2022-12-29T12:48:45.22128Z","iopub.status.idle":"2022-12-29T12:48:45.226692Z","shell.execute_reply.started":"2022-12-29T12:48:45.221243Z","shell.execute_reply":"2022-12-29T12:48:45.225506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.stat('/kaggle/input/nfl-player-contact-detection')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:45.244498Z","iopub.execute_input":"2022-12-29T12:48:45.245088Z","iopub.status.idle":"2022-12-29T12:48:45.252661Z","shell.execute_reply.started":"2022-12-29T12:48:45.245032Z","shell.execute_reply":"2022-12-29T12:48:45.251502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install folderstats","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:45.255108Z","iopub.execute_input":"2022-12-29T12:48:45.255663Z","iopub.status.idle":"2022-12-29T12:48:54.586081Z","shell.execute_reply.started":"2022-12-29T12:48:45.255625Z","shell.execute_reply":"2022-12-29T12:48:54.584782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import folderstats\nimport matplotlib.pyplot as plt \nimport squarify\nimport numpy as np\nimport networkx as nx\nfrom networkx.drawing.nx_pydot import graphviz_layout\n","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:54.587964Z","iopub.execute_input":"2022-12-29T12:48:54.588391Z","iopub.status.idle":"2022-12-29T12:48:54.595009Z","shell.execute_reply.started":"2022-12-29T12:48:54.58835Z","shell.execute_reply":"2022-12-29T12:48:54.593924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntest = folderstats.folderstats('/kaggle/input/nfl-player-contact-detection/test', ignore_hidden=True)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:54.597837Z","iopub.execute_input":"2022-12-29T12:48:54.599082Z","iopub.status.idle":"2022-12-29T12:48:54.627831Z","shell.execute_reply.started":"2022-12-29T12:48:54.599031Z","shell.execute_reply":"2022-12-29T12:48:54.626944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwith plt.style.context('ggplot'):\n    test = df['id'].value_counts().plot(\n        kind='bar', color='C1', title='Id Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:54.629056Z","iopub.execute_input":"2022-12-29T12:48:54.629854Z","iopub.status.idle":"2022-12-29T12:48:54.838826Z","shell.execute_reply.started":"2022-12-29T12:48:54.629816Z","shell.execute_reply":"2022-12-29T12:48:54.837974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by id and sum all sizes for each id \n    extension_sizes = test = df.groupby('id')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Id Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:54.840213Z","iopub.execute_input":"2022-12-29T12:48:54.840573Z","iopub.status.idle":"2022-12-29T12:48:55.063481Z","shell.execute_reply.started":"2022-12-29T12:48:54.840539Z","shell.execute_reply":"2022-12-29T12:48:55.062633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Group by id and sum all sizes for each id\ntest = test = df.groupby('id')['size'].sum()\n# Sort elments by size\ntest = id_sizes.sort_values(ascending=False)\n\nsquarify.plot(sizes=test.values, label=test.index.values)\nplt.title('Id Treemap by Size')\nplt.axis('off');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:55.064836Z","iopub.execute_input":"2022-12-29T12:48:55.065167Z","iopub.status.idle":"2022-12-29T12:48:55.175868Z","shell.execute_reply.started":"2022-12-29T12:48:55.065134Z","shell.execute_reply":"2022-12-29T12:48:55.174656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Filter the data set to only folders\n    test = test = df[df['folder']]\n    # Set the id to be the index (so we can use it as a label later)\n    test.set_index('id', inplace=True)\n    # Sort the folders by size\n    test = df_folders.sort_values(by='size', ascending=False)\n    \n    # Show the size of the largest 1 folder as a bar plot\n    test['size'][:50].plot(kind='bar', color='C0', title='Folder Sizes');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:55.180394Z","iopub.execute_input":"2022-12-29T12:48:55.180819Z","iopub.status.idle":"2022-12-29T12:48:55.39282Z","shell.execute_reply.started":"2022-12-29T12:48:55.180794Z","shell.execute_reply":"2022-12-29T12:48:55.391943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwith plt.style.context('ggplot'):\n    y, bins = np.histogram(df['id'], bins=10000)\n    plt.loglog(bins[:-1], y, '.');\n    plt.ylabel('Size')\n    plt.xlabel('Rank')\n","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:55.396595Z","iopub.execute_input":"2022-12-29T12:48:55.396865Z","iopub.status.idle":"2022-12-29T12:48:55.91555Z","shell.execute_reply.started":"2022-12-29T12:48:55.39684Z","shell.execute_reply":"2022-12-29T12:48:55.91461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Sort the index\ntest_sorted = df.sort_values(by='id')\n\nG = nx.Graph()\nfor i, row in df_sorted.iterrows():\n    if row.parent:\n        G.add_edge(row.id, row.parent)\n    \n# Print some additional information\nprint(nx.info(G))","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:55.917923Z","iopub.execute_input":"2022-12-29T12:48:55.918897Z","iopub.status.idle":"2022-12-29T12:48:55.927863Z","shell.execute_reply.started":"2022-12-29T12:48:55.918856Z","shell.execute_reply":"2022-12-29T12:48:55.926696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntest = pos_dot = graphviz_layout(G, prog='dot')\n\nfig = plt.figure(figsize=(16, 8))\nnodes = nx.draw_networkx_nodes(G, pos_dot, node_size=2, node_color='C0')\nedges = nx.draw_networkx_edges(G, pos_dot, edge_color='C0', width=0.5)\nplt.axis('off');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:55.929484Z","iopub.execute_input":"2022-12-29T12:48:55.930083Z","iopub.status.idle":"2022-12-29T12:48:56.19017Z","shell.execute_reply.started":"2022-12-29T12:48:55.930025Z","shell.execute_reply":"2022-12-29T12:48:56.189015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pos_twopi = graphviz_layout(G, prog='twopi', root=1)\n\nfig = plt.figure(figsize=(14, 14))\nnodes = nx.draw_networkx_nodes(G, pos_twopi, node_size=2, node_color='C0')\nedges = nx.draw_networkx_edges(G, pos_twopi, edge_color='C0', width=0.5)\nplt.axis('off')\nplt.axis('equal');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:56.192309Z","iopub.execute_input":"2022-12-29T12:48:56.192951Z","iopub.status.idle":"2022-12-29T12:48:56.474187Z","shell.execute_reply.started":"2022-12-29T12:48:56.192909Z","shell.execute_reply":"2022-12-29T12:48:56.472955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    df['path'].value_counts().plot(\n        kind='bar', color='C1', title='Path Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:56.476568Z","iopub.execute_input":"2022-12-29T12:48:56.477236Z","iopub.status.idle":"2022-12-29T12:48:56.746886Z","shell.execute_reply.started":"2022-12-29T12:48:56.477195Z","shell.execute_reply":"2022-12-29T12:48:56.74599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by path and sum all sizes for each path \n    extension_sizes = test = df.groupby('path')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Path Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:56.748739Z","iopub.execute_input":"2022-12-29T12:48:56.749763Z","iopub.status.idle":"2022-12-29T12:48:57.027839Z","shell.execute_reply.started":"2022-12-29T12:48:56.749726Z","shell.execute_reply":"2022-12-29T12:48:57.026935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group by path and sum all sizes for each path\ntest = df.groupby('path')['size'].sum()\n# Sort elments by size\ntest = id_sizes.sort_values(ascending=False)\n\nsquarify.plot(sizes=test.values, label=test.index.values)\nplt.title('Path Treemap by Size')\nplt.axis('off');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:57.029203Z","iopub.execute_input":"2022-12-29T12:48:57.030169Z","iopub.status.idle":"2022-12-29T12:48:57.141526Z","shell.execute_reply.started":"2022-12-29T12:48:57.030132Z","shell.execute_reply":"2022-12-29T12:48:57.140232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Filter the data set to only folders\n    test = df[df['folder']]\n    # Set the id to be the index (so we can use it as a label later)\n    test.set_index('path', inplace=True)\n    # Sort the folders by size\n    test = df_folders.sort_values(by='size', ascending=False)\n    \n    # Show the size of the largest 1 folder as a bar plot\n    test['size'][:50].plot(kind='bar', color='C0', title='Folder Sizes');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:57.143096Z","iopub.execute_input":"2022-12-29T12:48:57.143433Z","iopub.status.idle":"2022-12-29T12:48:57.358863Z","shell.execute_reply.started":"2022-12-29T12:48:57.143399Z","shell.execute_reply":"2022-12-29T12:48:57.357984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    test = df['name'].value_counts().plot(\n        kind='bar', color='C1', title='Name Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:57.359971Z","iopub.execute_input":"2022-12-29T12:48:57.360332Z","iopub.status.idle":"2022-12-29T12:48:57.663487Z","shell.execute_reply.started":"2022-12-29T12:48:57.360301Z","shell.execute_reply":"2022-12-29T12:48:57.662602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by name and sum all sizes for each name\n    extension_sizes = test = df.groupby('name')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Name Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:57.667594Z","iopub.execute_input":"2022-12-29T12:48:57.669706Z","iopub.status.idle":"2022-12-29T12:48:57.968635Z","shell.execute_reply.started":"2022-12-29T12:48:57.669667Z","shell.execute_reply":"2022-12-29T12:48:57.967839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n  test = df['extension'].value_counts().plot(\n        kind='bar', color='C1', title='Extension Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:57.972489Z","iopub.execute_input":"2022-12-29T12:48:57.974573Z","iopub.status.idle":"2022-12-29T12:48:58.174245Z","shell.execute_reply.started":"2022-12-29T12:48:57.974536Z","shell.execute_reply":"2022-12-29T12:48:58.173328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by extension and sum all sizes for each extension \n    extension_sizes = test = df.groupby('extension')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Extension Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:58.175641Z","iopub.execute_input":"2022-12-29T12:48:58.176139Z","iopub.status.idle":"2022-12-29T12:48:58.371913Z","shell.execute_reply.started":"2022-12-29T12:48:58.176098Z","shell.execute_reply":"2022-12-29T12:48:58.370966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    test = df['size'].value_counts().plot(\n        kind='bar', color='C1', title='Size Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:58.374251Z","iopub.execute_input":"2022-12-29T12:48:58.374856Z","iopub.status.idle":"2022-12-29T12:48:58.58404Z","shell.execute_reply.started":"2022-12-29T12:48:58.374815Z","shell.execute_reply":"2022-12-29T12:48:58.583122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by size and sum all sizes for each size\n    extension_sizes = tets = df.groupby('size')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Size Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:58.58556Z","iopub.execute_input":"2022-12-29T12:48:58.585917Z","iopub.status.idle":"2022-12-29T12:48:58.814833Z","shell.execute_reply.started":"2022-12-29T12:48:58.585882Z","shell.execute_reply":"2022-12-29T12:48:58.813946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwith plt.style.context('ggplot'):\n    y, bins = np.histogram(df['size'], bins=10000)\n    plt.loglog(bins[:-1], y, '.');\n    plt.ylabel('Size')\n    plt.xlabel('Rank')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:58.816159Z","iopub.execute_input":"2022-12-29T12:48:58.816759Z","iopub.status.idle":"2022-12-29T12:48:59.26991Z","shell.execute_reply.started":"2022-12-29T12:48:58.816723Z","shell.execute_reply":"2022-12-29T12:48:59.268989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    test = df['atime'].value_counts().plot(\n        kind='bar', color='C1', title='Atime Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:59.275958Z","iopub.execute_input":"2022-12-29T12:48:59.276271Z","iopub.status.idle":"2022-12-29T12:48:59.497614Z","shell.execute_reply.started":"2022-12-29T12:48:59.276244Z","shell.execute_reply":"2022-12-29T12:48:59.496765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by atime and sum all sizes for each atime\n    extension_sizes = test = df.groupby('atime')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Atime Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:59.498865Z","iopub.execute_input":"2022-12-29T12:48:59.5003Z","iopub.status.idle":"2022-12-29T12:48:59.738882Z","shell.execute_reply.started":"2022-12-29T12:48:59.500263Z","shell.execute_reply":"2022-12-29T12:48:59.738038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    test = df['mtime'].value_counts().plot(\n        kind='bar', color='C1', title='Mtime Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:59.740258Z","iopub.execute_input":"2022-12-29T12:48:59.740598Z","iopub.status.idle":"2022-12-29T12:48:59.952405Z","shell.execute_reply.started":"2022-12-29T12:48:59.740564Z","shell.execute_reply":"2022-12-29T12:48:59.951548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by mtime and sum all sizes for each mtime\n    extension_sizes = test = df.groupby('mtime')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Mtime Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:48:59.953789Z","iopub.execute_input":"2022-12-29T12:48:59.954144Z","iopub.status.idle":"2022-12-29T12:49:00.160307Z","shell.execute_reply.started":"2022-12-29T12:48:59.95411Z","shell.execute_reply":"2022-12-29T12:49:00.159483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    test = df['ctime'].value_counts().plot(\n        kind='bar', color='C1', title='Ctime Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:00.161682Z","iopub.execute_input":"2022-12-29T12:49:00.162021Z","iopub.status.idle":"2022-12-29T12:49:00.373989Z","shell.execute_reply.started":"2022-12-29T12:49:00.161987Z","shell.execute_reply":"2022-12-29T12:49:00.373002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by ctime and sum all sizes for each ctime\n    extension_sizes = test = df.groupby('ctime')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Ctime Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:00.375325Z","iopub.execute_input":"2022-12-29T12:49:00.376144Z","iopub.status.idle":"2022-12-29T12:49:00.581849Z","shell.execute_reply.started":"2022-12-29T12:49:00.376083Z","shell.execute_reply":"2022-12-29T12:49:00.580983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    test = df['folder'].value_counts().plot(\n        kind='bar', color='C1', title='Folder Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:00.583018Z","iopub.execute_input":"2022-12-29T12:49:00.583375Z","iopub.status.idle":"2022-12-29T12:49:00.76535Z","shell.execute_reply.started":"2022-12-29T12:49:00.583339Z","shell.execute_reply":"2022-12-29T12:49:00.764552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by folder and sum all sizes for each folder \n    extension_sizes = test = df.groupby('folder')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Folder Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:00.768441Z","iopub.execute_input":"2022-12-29T12:49:00.76871Z","iopub.status.idle":"2022-12-29T12:49:00.968666Z","shell.execute_reply.started":"2022-12-29T12:49:00.768684Z","shell.execute_reply":"2022-12-29T12:49:00.967789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwith plt.style.context('ggplot'):\n    y, bins = np.histogram(df['folder'], bins=10000)\n    plt.loglog(bins[:-1], y, '.');\n    plt.ylabel('Size')\n    plt.xlabel('Rank')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:00.969769Z","iopub.execute_input":"2022-12-29T12:49:00.970144Z","iopub.status.idle":"2022-12-29T12:49:01.984941Z","shell.execute_reply.started":"2022-12-29T12:49:00.970109Z","shell.execute_reply":"2022-12-29T12:49:01.984012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    test = df['num_files'].value_counts().plot(\n        kind='bar', color='C1', title='Numfiles Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:01.986581Z","iopub.execute_input":"2022-12-29T12:49:01.986967Z","iopub.status.idle":"2022-12-29T12:49:02.159888Z","shell.execute_reply.started":"2022-12-29T12:49:01.986931Z","shell.execute_reply":"2022-12-29T12:49:02.158966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by num_files and sum all sizes for each num_files \n    extension_sizes = test = df.groupby('num_files')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Numfiles Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:02.161358Z","iopub.execute_input":"2022-12-29T12:49:02.163948Z","iopub.status.idle":"2022-12-29T12:49:02.353909Z","shell.execute_reply.started":"2022-12-29T12:49:02.163918Z","shell.execute_reply":"2022-12-29T12:49:02.353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    test = df['depth'].value_counts().plot(\n        kind='bar', color='C1', title='Depth Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:02.35542Z","iopub.execute_input":"2022-12-29T12:49:02.35578Z","iopub.status.idle":"2022-12-29T12:49:02.534254Z","shell.execute_reply.started":"2022-12-29T12:49:02.355727Z","shell.execute_reply":"2022-12-29T12:49:02.53298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by depth and sum all sizes for each depth \n    extension_sizes = test = df.groupby('depth')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Depth Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:02.535695Z","iopub.execute_input":"2022-12-29T12:49:02.536016Z","iopub.status.idle":"2022-12-29T12:49:02.723919Z","shell.execute_reply.started":"2022-12-29T12:49:02.535982Z","shell.execute_reply":"2022-12-29T12:49:02.722977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwith plt.style.context('ggplot'):\n    y, bins = np.histogram(df['depth'], bins=10000)\n    plt.loglog(bins[:-1], y, '.');\n    plt.ylabel('Size')\n    plt.xlabel('Rank')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:02.725091Z","iopub.execute_input":"2022-12-29T12:49:02.725422Z","iopub.status.idle":"2022-12-29T12:49:03.267006Z","shell.execute_reply.started":"2022-12-29T12:49:02.725387Z","shell.execute_reply":"2022-12-29T12:49:03.265985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    test = df['parent'].value_counts().plot(\n        kind='bar', color='C1', title='Parent Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:03.268299Z","iopub.execute_input":"2022-12-29T12:49:03.268918Z","iopub.status.idle":"2022-12-29T12:49:03.449854Z","shell.execute_reply.started":"2022-12-29T12:49:03.26888Z","shell.execute_reply":"2022-12-29T12:49:03.449005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by parent and sum all sizes for each parent\n    extension_sizes = test = df.groupby('parent')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Parent Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:03.45108Z","iopub.execute_input":"2022-12-29T12:49:03.45144Z","iopub.status.idle":"2022-12-29T12:49:03.652917Z","shell.execute_reply.started":"2022-12-29T12:49:03.451389Z","shell.execute_reply":"2022-12-29T12:49:03.652003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwith plt.style.context('ggplot'):\n    y, bins = np.histogram(df['parent'], bins=10000)\n    plt.loglog(bins[:-1], y, '.');\n    plt.ylabel('Size')\n    plt.xlabel('Rank')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:03.654116Z","iopub.execute_input":"2022-12-29T12:49:03.65498Z","iopub.status.idle":"2022-12-29T12:49:04.268001Z","shell.execute_reply.started":"2022-12-29T12:49:03.654943Z","shell.execute_reply":"2022-12-29T12:49:04.267034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    test = df['uid'].value_counts().plot(\n        kind='bar', color='C1', title='Uid Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:04.269566Z","iopub.execute_input":"2022-12-29T12:49:04.270049Z","iopub.status.idle":"2022-12-29T12:49:04.454349Z","shell.execute_reply.started":"2022-12-29T12:49:04.27001Z","shell.execute_reply":"2022-12-29T12:49:04.453166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by uid and sum all sizes for each uid \n    extension_sizes = test = df.groupby('uid')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Uid Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:04.461209Z","iopub.execute_input":"2022-12-29T12:49:04.463246Z","iopub.status.idle":"2022-12-29T12:49:04.651847Z","shell.execute_reply.started":"2022-12-29T12:49:04.463217Z","shell.execute_reply":"2022-12-29T12:49:04.650973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwith plt.style.context('ggplot'):\n    y, bins = np.histogram(df['uid'], bins=10000)\n    plt.loglog(bins[:-1], y, '.');\n    plt.ylabel('Size')\n    plt.xlabel('Rank')","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:04.65342Z","iopub.execute_input":"2022-12-29T12:49:04.653744Z","iopub.status.idle":"2022-12-29T12:49:05.09291Z","shell.execute_reply.started":"2022-12-29T12:49:04.653709Z","shell.execute_reply":"2022-12-29T12:49:05.092017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain = folderstats.folderstats('/kaggle/input/nfl-player-contact-detection/train', ignore_hidden=True)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:05.095894Z","iopub.execute_input":"2022-12-29T12:49:05.096207Z","iopub.status.idle":"2022-12-29T12:49:05.376615Z","shell.execute_reply.started":"2022-12-29T12:49:05.09618Z","shell.execute_reply":"2022-12-29T12:49:05.375486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    train = df['id'].value_counts().plot(\n        kind='bar', color='C1', title='Id Distribution by Count');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:05.378109Z","iopub.execute_input":"2022-12-29T12:49:05.378453Z","iopub.status.idle":"2022-12-29T12:49:05.585403Z","shell.execute_reply.started":"2022-12-29T12:49:05.378418Z","shell.execute_reply":"2022-12-29T12:49:05.584593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Group by id and sum all sizes for each id \n    extension_sizes = train = df.groupby('id')['size'].sum()\n    # Sort elements by size\n    extension_sizes = extension_sizes.sort_values(ascending=False)\n    \n    extension_sizes.plot(\n        kind='bar', color='C1', title='Id Distribution by Size');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:05.586943Z","iopub.execute_input":"2022-12-29T12:49:05.587326Z","iopub.status.idle":"2022-12-29T12:49:05.806882Z","shell.execute_reply.started":"2022-12-29T12:49:05.587282Z","shell.execute_reply":"2022-12-29T12:49:05.806038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group by id and sum all sizes for each id\ntrain = train = df.groupby('id')['size'].sum()\n# Sort elments by size\ntrain = id_sizes.sort_values(ascending=False)\n\nsquarify.plot(sizes=train.values, label=train.index.values)\nplt.title('Id Treemap by Size')\nplt.axis('off');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:05.808156Z","iopub.execute_input":"2022-12-29T12:49:05.808499Z","iopub.status.idle":"2022-12-29T12:49:05.919509Z","shell.execute_reply.started":"2022-12-29T12:49:05.808462Z","shell.execute_reply":"2022-12-29T12:49:05.918355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with plt.style.context('ggplot'):\n    # Filter the data set to only folders\n    train = train = df[df['folder']]\n    # Set the id to be the index (so we can use it as a label later)\n    train.set_index('id', inplace=True)\n    # Sort the folders by size\n    train = df_folders.sort_values(by='size', ascending=False)\n    \n    # Show the size of the largest 1 folder as a bar plot\n    train['size'][:50].plot(kind='bar', color='C0', title='Folder Sizes');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:05.92533Z","iopub.execute_input":"2022-12-29T12:49:05.925889Z","iopub.status.idle":"2022-12-29T12:49:06.137214Z","shell.execute_reply.started":"2022-12-29T12:49:05.925838Z","shell.execute_reply":"2022-12-29T12:49:06.136358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwith plt.style.context('ggplot'):\n    y, bins = np.histogram(df['id'], bins=10000)\n    plt.loglog(bins[:-1], y, '.');\n    plt.ylabel('Size')\n    plt.xlabel('Rank')\n","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:06.138923Z","iopub.execute_input":"2022-12-29T12:49:06.139573Z","iopub.status.idle":"2022-12-29T12:49:06.648001Z","shell.execute_reply.started":"2022-12-29T12:49:06.139535Z","shell.execute_reply":"2022-12-29T12:49:06.64706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sort the index\ntrain_sorted = df.sort_values(by='id')\n\nG = nx.Graph()\nfor i, row in df_sorted.iterrows():\n    if row.parent:\n        G.add_edge(row.id, row.parent)\n    \n# Print some additional information\nprint(nx.info(G))","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:06.649344Z","iopub.execute_input":"2022-12-29T12:49:06.650324Z","iopub.status.idle":"2022-12-29T12:49:06.66042Z","shell.execute_reply.started":"2022-12-29T12:49:06.650284Z","shell.execute_reply":"2022-12-29T12:49:06.659462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain = pos_dot = graphviz_layout(G, prog='dot')\n\nfig = plt.figure(figsize=(16, 8))\nnodes = nx.draw_networkx_nodes(G, pos_dot, node_size=2, node_color='C0')\nedges = nx.draw_networkx_edges(G, pos_dot, edge_color='C0', width=0.5)\nplt.axis('off');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:06.6619Z","iopub.execute_input":"2022-12-29T12:49:06.662404Z","iopub.status.idle":"2022-12-29T12:49:06.947765Z","shell.execute_reply.started":"2022-12-29T12:49:06.662367Z","shell.execute_reply":"2022-12-29T12:49:06.946175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pos_twopi = graphviz_layout(G, prog='twopi', root=1)\n\nfig = plt.figure(figsize=(14, 14))\nnodes = nx.draw_networkx_nodes(G, pos_twopi, node_size=2, node_color='C0')\nedges = nx.draw_networkx_edges(G, pos_twopi, edge_color='C0', width=0.5)\nplt.axis('off')\nplt.axis('equal');","metadata":{"execution":{"iopub.status.busy":"2022-12-29T12:49:06.94972Z","iopub.execute_input":"2022-12-29T12:49:06.954763Z","iopub.status.idle":"2022-12-29T12:49:07.332652Z","shell.execute_reply.started":"2022-12-29T12:49:06.954316Z","shell.execute_reply":"2022-12-29T12:49:07.331594Z"},"trusted":true},"execution_count":null,"outputs":[]}]}