{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":105741,"databundleVersionId":12799113,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install fireducks\n!pip install pyarrow>=12.0.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T03:54:03.979755Z","iopub.execute_input":"2025-06-24T03:54:03.980561Z","iopub.status.idle":"2025-06-24T03:54:17.298186Z","shell.execute_reply.started":"2025-06-24T03:54:03.980528Z","shell.execute_reply":"2025-06-24T03:54:17.297057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import fireducks.pandas as pd\nimport numpy as np","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-24T03:54:22.728699Z","iopub.execute_input":"2025-06-24T03:54:22.729047Z","iopub.status.idle":"2025-06-24T03:54:24.575462Z","shell.execute_reply.started":"2025-06-24T03:54:22.729004Z","shell.execute_reply":"2025-06-24T03:54:24.574497Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def quick_eda(file_path):\n    \"\"\"\n    EDA nhanh với các thống kê cơ bản\n    \"\"\"\n    # Đọc dữ liệu\n    df = pd.read_csv(file_path)\n    \n    # Lấy cột số\n    numeric_cols = df.select_dtypes(include=[np.number]).columns.tolist()\n    \n    print(f\"Dataset shape: {df.shape}\")\n    print(f\"Numeric columns: {numeric_cols}\")\n    \n    # Thống kê cho từng cột\n    stats = {}\n    for col in numeric_cols:\n        stats[col] = {\n            'min': df[col].min(),\n            'max': df[col].max(),\n            'mean': df[col].mean(),\n            'median': df[col].median(),\n            '25%': df[col].quantile(0.25),\n            '75%': df[col].quantile(0.75)\n        }\n    \n    # Ma trận tương quan\n    correlation_matrix = df[numeric_cols].corr() if len(numeric_cols) > 1 else None\n    \n    # In kết quả\n    print(f\"\\nSTATISTICS:\")\n    for col, stat in stats.items():\n        print(f\"\\n{col}:\")\n        for key, value in stat.items():\n            print(f\"  {key}: {value:.4f}\")\n    \n    if correlation_matrix is not None:\n        print(f\"\\nORRELATION MATRIX:\")\n        print(correlation_matrix)\n    \n    return {'stats': stats, 'correlation': correlation_matrix}\n\n# Sử dụng\nresult = quick_eda(\"/kaggle/input/fds-summer-challenge-2025-weather-prediction/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-24T03:57:02.547922Z","iopub.execute_input":"2025-06-24T03:57:02.548501Z","iopub.status.idle":"2025-06-24T03:58:29.972862Z","shell.execute_reply.started":"2025-06-24T03:57:02.548470Z","shell.execute_reply":"2025-06-24T03:58:29.971542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}