{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":35332,"databundleVersionId":3723648},{"sourceType":"datasetVersion","sourceId":3739819,"datasetId":2231132,"databundleVersionId":3794269}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport dask.dataframe as dd\nimport matplotlib.pyplot as plt\nimport gc\nimport os\nimport time\nprint(\"Librerias importadas\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-27T15:39:31.780299Z","iopub.execute_input":"2024-11-27T15:39:31.780736Z","iopub.status.idle":"2024-11-27T15:39:33.603223Z","shell.execute_reply.started":"2024-11-27T15:39:31.780700Z","shell.execute_reply":"2024-11-27T15:39:33.601691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!conda install dask distributed -c conda-forge -y","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T15:39:33.605892Z","iopub.execute_input":"2024-11-27T15:39:33.606746Z","iopub.status.idle":"2024-11-27T15:41:58.765366Z","shell.execute_reply.started":"2024-11-27T15:39:33.606686Z","shell.execute_reply":"2024-11-27T15:41:58.763657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from dask.distributed import LocalCluster, Client\n\nnworkers = 10 # num workers\nnthreads = 2 # hilos por worker\n\ncluster = LocalCluster(n_workers=nworkers, threads_per_worker=nthreads)\nclient = Client(cluster)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T16:00:48.791003Z","iopub.execute_input":"2024-11-27T16:00:48.791516Z","iopub.status.idle":"2024-11-27T16:01:00.932827Z","shell.execute_reply.started":"2024-11-27T16:00:48.791471Z","shell.execute_reply":"2024-11-27T16:01:00.927357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## ANALISIS DE ESCALABILIDAD\nworkers_threads_start_time = time.perf_counter()\n# blocksize controla el tamaño de cada partición del DataFrame. Es importante elegir un tamaño adecuado para un rendimiento óptimo.\ncategorical_columns = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\ncat_df = dd.read_csv(\"/kaggle/input/amex-default-prediction/train_data.csv\", usecols=categorical_columns[:2])\ncat_df = cat_df.compute()\nworkers_threads_end_time = time.perf_counter()\nelapsed_WT = workers_threads_end_time - workers_threads_start_time\nprint(f\"Tiempo con {nworkers} workers y {nthreads} hilos: {elapsed_WT:.4f} segundos\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T16:01:09.096925Z","iopub.execute_input":"2024-11-27T16:01:09.098045Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}