{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Purpose\nHopefully this notebook will help discover feature characteristics that could be used to derive new features or select the best training data.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom pathlib import Path\nfrom IPython.display import display\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\npd.options.mode.copy_on_write = True\npd.set_option(\"display.max_columns\", None)\npd.set_option(\"display.max_rows\", 100)\n\npath = Path(\"/kaggle/input/jane-street-real-time-market-data-forecasting\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T12:12:29.989185Z","iopub.execute_input":"2024-12-02T12:12:29.989635Z","iopub.status.idle":"2024-12-02T12:12:31.296848Z","shell.execute_reply.started":"2024-12-02T12:12:29.989594Z","shell.execute_reply":"2024-12-02T12:12:31.295579Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Available Data Files","metadata":{}},{"cell_type":"code","source":"sorted_items = sorted(path.iterdir())\nfor item in sorted_items:\n    print(item)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T12:13:00.267025Z","iopub.execute_input":"2024-12-02T12:13:00.267607Z","iopub.status.idle":"2024-12-02T12:13:00.275309Z","shell.execute_reply.started":"2024-12-02T12:13:00.267571Z","shell.execute_reply":"2024-12-02T12:13:00.274189Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for file in path.rglob(\"date_id=*/*\"):\n    if file.is_file():\n        print(file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T12:15:28.491256Z","iopub.execute_input":"2024-12-02T12:15:28.491637Z","iopub.status.idle":"2024-12-02T12:15:28.526005Z","shell.execute_reply.started":"2024-12-02T12:15:28.491604Z","shell.execute_reply":"2024-12-02T12:15:28.524889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for file in path.rglob(\"partition_id=*/*\"):\n    if file.is_file():\n        print(f\"{file.parent.name}/{file.name}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T12:15:28.944540Z","iopub.execute_input":"2024-12-02T12:15:28.944920Z","iopub.status.idle":"2024-12-02T12:15:28.983266Z","shell.execute_reply.started":"2024-12-02T12:15:28.944888Z","shell.execute_reply":"2024-12-02T12:15:28.981833Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Display Functions\nGeneral helper functions to produce correlation maps and histograms.","metadata":{}},{"cell_type":"code","source":"def plot_correlation_heatmap(corr_matrix, figsize=(32, 20), title=\"Correlation Heatmap\", cmap=\"coolwarm\"):\n    plt.figure(figsize=figsize)\n    sns.heatmap(corr_matrix, annot=True, fmt=\".2f\", cmap=cmap, annot_kws={\"size\": 6})\n    plt.title(title, fontsize=20)\n    plt.xticks(fontsize=12)\n    plt.yticks(fontsize=12)\n    plt.show()\n\n\ndef plot_histograms(df, bins=30, figsize=(30, 20), title=\"Histograms for Features\"):\n    df.hist(bins=bins, figsize=figsize)\n    plt.suptitle(title, fontsize=20)\n    plt.subplots_adjust(hspace=0.4, wspace=0.4)\n    plt.show()\n\n\ndef display_correlation_features(corr_matrix, title=\"Correlation Spread Features\"):\n    print(f\"\\n{title}\")\n    print(\"Features with high correlation (around 1.0) for all features:\")\n    high_corr_all = corr_matrix[(corr_matrix >= 0.9) & (corr_matrix <= 1.0)].stack().index.tolist()\n    display(get_unique_combinations(high_corr_all))\n\n    print(\"Features with low correlation (around 0) for all features:\")\n    low_corr_all = corr_matrix[(corr_matrix >= -0.1) & (corr_matrix <= 0.1)].stack().index.tolist()\n    display(get_unique_combinations(low_corr_all))\n\n    print(\"Features with high negative correlation (around -1.0) for all features:\")\n    neg_high_corr_all = corr_matrix[(corr_matrix <= -0.9) & (corr_matrix >= -1.0)].stack().index.tolist()\n    display(get_unique_combinations(neg_high_corr_all))\n\n\ndef display_correlation_features_above_threshold(high_corr_features, threshold, partition_id):\n    for feature in high_corr_features:\n        unique_items = get_unique_items(feature, high_corr_features)\n        print(f\"partition_ids: {partition_id}\")\n        print(f\"{feature} correlation >= {threshold}\")\n        display(unique_items)\n\n\ndef get_unique_combinations(tuples_list):\n    unique_combinations = set()\n    for a, b in tuples_list:\n        if a != b:\n            unique_combinations.add(tuple(sorted((a, b))))\n    return list(unique_combinations)\n\n\ndef get_unique_items(feature, high_corr_features):\n    if feature in high_corr_features:\n        unique_items = list(set(high_corr_features[feature]))\n        return unique_items\n    else:\n        return []","metadata":{"trusted":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-02T12:16:33.765888Z","iopub.execute_input":"2024-12-02T12:16:33.767398Z","iopub.status.idle":"2024-12-02T12:16:33.780713Z","shell.execute_reply.started":"2024-12-02T12:16:33.767337Z","shell.execute_reply":"2024-12-02T12:16:33.779398Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Processor","metadata":{}},{"cell_type":"code","source":"class DataProcessor:\n    def __init__(self, path, partitions, selected_features, corr_above_threshold=0.3):\n        self.path = path\n        self.partitions = self.process_partitions(partitions)\n        self.selected_features = selected_features\n        self.shapes = []\n        self.columns_list = []\n        self.unique_summary = []\n        self.corr_matrices = {}\n        self.dataframes = {}\n        self.high_corr_features = {}\n        self.corr_above_threshold = corr_above_threshold\n\n    def process_partitions(self, partitions):\n        if isinstance(partitions, int):\n            return list(range(partitions))\n        elif isinstance(partitions, tuple) and len(partitions) == 2:\n            return list(range(partitions[0], partitions[1] + 1))\n        elif isinstance(partitions, list):\n            return partitions\n        elif isinstance(partitions, slice):\n            return list(range(partitions.start, partitions.stop, partitions.step or 1))\n        else:\n            raise ValueError(\"Partitions must be an integer, a tuple of two integers, a list, or a slice\")\n\n    def process_file(self, file_path, partition_id, display_summary=True):\n        print(f\"Processing file: {file_path} for partition_id: {partition_id}\")\n        df = pd.read_parquet(file_path)\n        self.dataframes[partition_id] = df\n\n        self.shapes.append(df.shape)\n        self.columns_list.append(df.columns.to_list())\n\n        unique_data = {feature: df[feature].nunique() for feature in self.selected_features}\n        self.unique_summary.append({\"File ID\": f\"File id={partition_id}\", **unique_data})\n\n        corr_matrix = df.corr()\n        self.corr_matrices[partition_id] = corr_matrix\n\n        features_to_consider = (\n            self.selected_features if self.selected_features else self.dataframes[partition_id].columns.tolist()\n        )\n        for feature in features_to_consider:\n            positive_corr = corr_matrix[corr_matrix[feature] >= self.corr_above_threshold][feature].index.tolist()\n            positive_corr = [f for f in positive_corr if f not in features_to_consider]\n\n            if feature not in self.high_corr_features:\n                self.high_corr_features[feature] = []\n\n            self.high_corr_features[feature].extend(positive_corr)\n\n        if display_summary:\n            display(df.head(3))\n            display(df.tail(3))\n            print(f\"Shape: {df.shape}\")\n\n            for feature, unique_count in unique_data.items():\n                print(f\"Unique {feature}: {unique_count}\")\n            print()\n\n    def summarize(self):\n        shape_df = pd.DataFrame(\n            self.shapes, columns=[\"Rows\", \"Columns\"], index=[f\"File id={i}\" for i in self.partitions]\n        )\n        print(\"Shape Comparison:\")\n        display(shape_df)\n        shape_df.plot(kind=\"bar\", figsize=(12, 6))\n        plt.title(f\"Shape Comparison for {self.partitions} File(s)\")\n        plt.xlabel(\"File ID\")\n        plt.ylabel(\"Count\")\n        plt.legend().set_visible(False)\n        plt.show()\n\n        unique_summary_df = pd.DataFrame(self.unique_summary)\n        print(\"Unique Summary:\")\n        display(unique_summary_df)\n        display_correlation_features_above_threshold(\n            self.high_corr_features, self.corr_above_threshold, self.partitions\n        )\n\n    def display_correlation_features_above_threshold(self):\n        for partition_id in self.partitions:\n            if partition_id in self.corr_matrices:\n                display_correlation_features_above_threshold(\n                    self.high_corr_features, self.corr_above_threshold, partition_id\n                )\n\n    def plot_correlation_heatmaps(self):\n        for partition_id in self.partitions:\n            if partition_id in self.corr_matrices:\n                corr_matrix = self.corr_matrices[partition_id]\n                plot_correlation_heatmap(corr_matrix, title=f\"Correlation Heatmap for Partition {partition_id}\")\n\n    def plot_selected_features_correlation_maps(self):\n        for partition_id in self.partitions:\n            if partition_id in self.corr_matrices:\n                corr_matrix = self.corr_matrices[partition_id]\n                correlation_map = corr_matrix.loc[self.selected_features, :]\n                correlation_map = correlation_map[correlation_map > 0].dropna(axis=1, how=\"all\")\n                plot_correlation_heatmap(\n                    correlation_map,\n                    figsize=(32, 4),\n                    title=f\"Selected Features Correlation Map for Partition {partition_id}\",\n                )\n\n    def display_correlation_features(self):\n        for partition_id in self.partitions:\n            if partition_id in self.corr_matrices:\n                corr_matrix = self.corr_matrices[partition_id]\n                display_correlation_features(\n                    corr_matrix, title=f\"Correlation Spread Features for Partition {partition_id}\"\n                )\n\n    def plot_histograms(self):\n        for partition_id in self.partitions:\n            if partition_id in self.dataframes:\n                df = self.dataframes[partition_id]\n                plot_histograms(df, title=f\"Histograms for Partition {partition_id}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T12:17:03.967959Z","iopub.execute_input":"2024-12-02T12:17:03.968336Z","iopub.status.idle":"2024-12-02T12:17:03.997518Z","shell.execute_reply.started":"2024-12-02T12:17:03.968304Z","shell.execute_reply":"2024-12-02T12:17:03.995983Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Process partitions 3 and 9\nNumber of symbols differ but date and time ids are similar.","metadata":{}},{"cell_type":"code","source":"analyzer = DataProcessor(path, partitions=[3, 9], selected_features=[\"date_id\", \"time_id\", \"symbol_id\"])\n\nfor partition_id in analyzer.partitions:\n    file_path = path / f\"train.parquet/partition_id={partition_id}/part-0.parquet\"\n    analyzer.process_file(file_path, partition_id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T12:17:20.399693Z","iopub.execute_input":"2024-12-02T12:17:20.400192Z","iopub.status.idle":"2024-12-02T12:22:05.774502Z","shell.execute_reply.started":"2024-12-02T12:17:20.400154Z","shell.execute_reply":"2024-12-02T12:22:05.772782Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Correlation Heatmaps & Histograms\nThere are a few feature groups highly correlated.  These seem to be similar groups for both partions.  \nThere seem to be some features somewhat correlated to symbols, date, and time.","metadata":{}},{"cell_type":"code","source":"analyzer.plot_correlation_heatmaps()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T12:22:05.777238Z","iopub.execute_input":"2024-12-02T12:22:05.777715Z","iopub.status.idle":"2024-12-02T12:22:40.050308Z","shell.execute_reply.started":"2024-12-02T12:22:05.777670Z","shell.execute_reply":"2024-12-02T12:22:40.048362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyzer.plot_selected_features_correlation_maps()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyzer.plot_histograms()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Correlation across all features\nAround 1, 0, -1. Consult display_correlation_features() for any adjustments.","metadata":{}},{"cell_type":"code","source":"analyzer.display_correlation_features()","metadata":{"trusted":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Summary over two training files","metadata":{}},{"cell_type":"code","source":"analyzer.summarize()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# All Partitions","metadata":{}},{"cell_type":"code","source":"analyzer = DataProcessor(path, partitions=10, selected_features=[\"date_id\", \"time_id\", \"symbol_id\"])\n\nfor partition_id in analyzer.partitions:\n    file_path = path / f\"train.parquet/partition_id={partition_id}/part-0.parquet\"\n    analyzer.process_file(file_path, partition_id)  # set display_summary=False to suppress output","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyzer.summarize()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyzer.plot_correlation_heatmaps()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyzer.plot_selected_features_correlation_maps()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyzer.display_correlation_features()","metadata":{"trusted":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyzer.display_correlation_features_above_threshold()","metadata":{"trusted":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyzer.plot_histograms()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}