{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# open problems data view","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-09-13T12:48:41.540045Z","iopub.execute_input":"2023-09-13T12:48:41.540568Z","iopub.status.idle":"2023-09-13T12:48:41.546037Z","shell.execute_reply.started":"2023-09-13T12:48:41.540525Z","shell.execute_reply":"2023-09-13T12:48:41.545029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths=[]\npaths2=[]\nfor dirname, _, filenames in os.walk('/kaggle/input/open-problems-single-cell-perturbations'):\n    for filename in filenames:\n        if filename[-4:]=='.csv':\n            paths+=[(os.path.join(dirname,filename))]\n        elif filename[-8:]=='.parquet':\n            paths2+=[(os.path.join(dirname,filename))]\n","metadata":{"execution":{"iopub.status.busy":"2023-09-13T12:48:41.548202Z","iopub.execute_input":"2023-09-13T12:48:41.548955Z","iopub.status.idle":"2023-09-13T12:48:47.723989Z","shell.execute_reply.started":"2023-09-13T12:48:41.548916Z","shell.execute_reply":"2023-09-13T12:48:47.722851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs=paths.copy()\nfor i,path in enumerate(paths):\n    dfs[i]=pd.read_csv(path)\n    print(path.split('/')[-1])\n    display(dfs[i])\n    print()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install dask","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import dask\nimport dask.dataframe as dd","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pqs=paths2.copy()\nfor i,path in enumerate(paths2):\n    pqs[i]=dd.read_parquet(path)\n    print(path.split('/')[-1])\n    display(pqs[i].iloc[:,:])\n    display(pqs[i].compute())\n    print()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}