{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":10245336,"sourceType":"datasetVersion","datasetId":6336300}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-29T10:58:48.880532Z","iopub.execute_input":"2024-12-29T10:58:48.880942Z","iopub.status.idle":"2024-12-29T10:58:50.340825Z","shell.execute_reply.started":"2024-12-29T10:58:48.880897Z","shell.execute_reply":"2024-12-29T10:58:50.339656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport polars as pl\n\nimport kaggle_evaluation.jane_street_inference_server\ntrain = pl.scan_parquet(\"/kaggle/input/20241219-data/training.parquet\").collect().to_pandas()\nvalid = pl.scan_parquet(\"/kaggle/input/20241219-data/validation.parquet\").collect().to_pandas()\ntrain.shape, valid.shape","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['feature_09']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T11:07:44.427633Z","iopub.execute_input":"2024-12-29T11:07:44.428082Z","iopub.status.idle":"2024-12-29T11:07:44.436719Z","shell.execute_reply.started":"2024-12-29T11:07:44.428046Z","shell.execute_reply":"2024-12-29T11:07:44.435285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"min_value = train['feature_09'].min()\nmax_value = train['feature_09'].max()\nprint(f\"Minimum value: {min_value}\")\nprint(f\"Maximum value: {max_value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T11:09:12.817951Z","iopub.execute_input":"2024-12-29T11:09:12.818462Z","iopub.status.idle":"2024-12-29T11:09:12.827147Z","shell.execute_reply.started":"2024-12-29T11:09:12.818417Z","shell.execute_reply":"2024-12-29T11:09:12.826009Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(10, 6))\nplt.hist(train_1['feature_09'], bins=40, color='yellow')\nplt.xlabel('feature09', fontsize=20)\nplt.ylabel('Frequency', fontsize=20)\nplt.grid(axis='y')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T11:11:36.982894Z","iopub.execute_input":"2024-12-29T11:11:36.983451Z","iopub.status.idle":"2024-12-29T11:11:37.356327Z","shell.execute_reply.started":"2024-12-29T11:11:36.983408Z","shell.execute_reply":"2024-12-29T11:11:37.354566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nplt.boxplot(train_1['feature_09'], vert=False, patch_artist=True)\nplt.xlabel('feature09', fontsize=20)\nplt.grid(axis='y')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T11:12:46.762258Z","iopub.execute_input":"2024-12-29T11:12:46.762672Z","iopub.status.idle":"2024-12-29T11:12:47.65921Z","shell.execute_reply.started":"2024-12-29T11:12:46.762636Z","shell.execute_reply":"2024-12-29T11:12:47.658107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport polars as pl\nimport matplotlib.pyplot as plt\n\n\ndef read_parquet_files(train_path: str, valid_path: str) -> (pd.DataFrame, pd.DataFrame):\n    \"\"\"\n    读取Parquet文件并转换为pandas DataFrame\n    \n    :param train_path: 训练集Parquet文件路径\n    :param valid_path: 验证集Parquet文件路径\n    :return: 包含训练集和验证集的DataFrame的元组\n    \"\"\"\n    try:\n        train = pl.scan_parquet(train_path).collect().to_pandas()\n        valid = pl.scan_parquet(valid_path).collect().to_pandas()\n        return train, valid\n    except FileNotFoundError as e:\n        print(f\"文件未找到: {e}\")\n        return None, None\n    except Exception as e:\n        print(f\"读取文件时发生错误: {e}\")\n        return None, None\n\n\ndef visualize_feature_09(data: pd.DataFrame, feature_name: str):\n    \"\"\"\n    对指定特征进行直方图和箱线图可视化\n    \n    :param data: 包含数据的DataFrame\n    :param feature_name: 要可视化的特征名称\n    \"\"\"\n    if data is None:\n        return\n    \n    # 直方图\n    plt.figure(figsize=(10, 6))\n    plt.hist(data[feature_name], bins=40, color='yellow')\n    plt.xlabel(f'{feature_name}', fontsize=20)\n    plt.ylabel('Frequency', fontsize=20)\n    plt.grid(axis='y')\n    plt.title(f'Histogram of {feature_name}')\n    plt.show()\n\n    # 箱线图\n    plt.figure(figsize=(10, 6))\n    plt.boxplot(data[feature_name], vert=False, patch_artist=True)\n    plt.xlabel(f'{feature_name}', fontsize=20)\n    plt.grid(axis='y')\n    plt.title(f'Boxplot of {feature_name}')\n    plt.show()\n\n\nif __name__ == \"__main__\":\n    train_path = \"/kaggle/input/20241219-data/training.parquet\"\n    valid_path = \"/kaggle/input/20241219-data/validation.parquet\"\n    \n    train, valid = read_parquet_files(train_path, valid_path)\n    if train is not None and valid is not None:\n        visualize_feature_09(train, 'feature_09')\n        visualize_feature_09(valid, 'feature_09')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-29T11:32:55.589861Z","iopub.execute_input":"2024-12-29T11:32:55.590352Z","iopub.status.idle":"2024-12-29T11:33:07.982288Z","shell.execute_reply.started":"2024-12-29T11:32:55.590318Z","shell.execute_reply":"2024-12-29T11:33:07.981043Z"}},"outputs":[],"execution_count":null}]}