{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:03:34.855816Z","iopub.execute_input":"2024-12-01T17:03:34.856220Z","iopub.status.idle":"2024-12-01T17:03:34.909600Z","shell.execute_reply.started":"2024-12-01T17:03:34.856187Z","shell.execute_reply":"2024-12-01T17:03:34.908294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd \nimport numpy as np\nimport pyarrow as pa\nmy_file_path = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=8/part-0.parquet\"\ndf1 = pd.read_parquet( my_file_path, engine = 'auto')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:03:39.937805Z","iopub.execute_input":"2024-12-01T17:03:39.938216Z","iopub.status.idle":"2024-12-01T17:03:50.986409Z","shell.execute_reply.started":"2024-12-01T17:03:39.938181Z","shell.execute_reply":"2024-12-01T17:03:50.985326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndef check_time_id_counts(data):\n    \"\"\"\n    Checks if each unique date_id has the same number of time_id values.\n    If true, prints the number of time_id elements.\n\n    Parameters:\n    - data: pandas DataFrame containing `date_id` and `time_i` columns.\n\n    Returns:\n    - None\n    \"\"\"\n\n    # Group by date_id and count unique time_id values\n    time_id_counts = data.groupby('date_id')['time_id'].nunique()\n\n    # Check if all counts are equal\n    if time_id_counts.nunique() == 1:\n        print(\"All date_id values have the same number of time_id values.\")\n        print(f\"Number of time_id values for each date_id: {time_id_counts.iloc[0]}\")\n    else:\n        print(\"Inconsistency detected! Not all date_id values have the same number of time_id values.\")\n        print(time_id_counts.reset_index(name='time_id_count'))\n\n# Example Usage\n\ncheck_time_id_counts(df1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:04:02.216606Z","iopub.execute_input":"2024-12-01T17:04:02.217053Z","iopub.status.idle":"2024-12-01T17:04:02.466633Z","shell.execute_reply.started":"2024-12-01T17:04:02.217014Z","shell.execute_reply":"2024-12-01T17:04:02.465439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport plotly.graph_objects as go\nfrom scipy.fft import fft, fftfreq\n\ndef process_and_plot(data, features_to_plot):\n    \"\"\"\n    Processes the data by merging `date_id` and `time_id`, plots selected features interactively,\n    and computes the Fourier Transform of the selected features.\n\n    Parameters:\n    - data: pandas DataFrame containing the data with `date_id`, `time_id`, and features.\n    - features_to_plot: List of features to plot and analyze.\n\n    Returns:\n    - fft_results: Dictionary with Fourier Transform results for each feature.\n    \"\"\"\n\n    # 1. Merge date_id and time_id to create a composite time column\n    # Ensure date_id is treated as days and time_id as intervals within the day\n    data['datetime'] = data['date_id'] + (data['time_id'] / data['time_id'].max())\n    data = data.sort_values(['date_id', 'time_id']).reset_index(drop=True)\n\n    # Plotting function using Plotly Graph Objects\n    def plot_features():\n        fig = go.Figure()\n        for feature in features_to_plot:\n            if feature in data.columns:\n                fig.add_trace(go.Scatter(\n                    x=data['datetime'],\n                    y=data[feature],\n                    mode='lines',\n                    name=feature\n                ))\n            else:\n                print(f\"Feature {feature} not found in data.\")\n        fig.update_layout(\n            title=\"Interactive Feature Plot\",\n            xaxis_title=\"Datetime (combined date_id + time_id)\",\n            yaxis_title=\"Feature Value\",\n            template=\"plotly_white\"\n        )\n        fig.show()\n\n    # 2. Plot the features\n    plot_features()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:04:07.115937Z","iopub.execute_input":"2024-12-01T17:04:07.116318Z","iopub.status.idle":"2024-12-01T17:04:07.125419Z","shell.execute_reply.started":"2024-12-01T17:04:07.116288Z","shell.execute_reply":"2024-12-01T17:04:07.124166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df1.describe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T16:49:15.643886Z","iopub.execute_input":"2024-12-01T16:49:15.644253Z","iopub.status.idle":"2024-12-01T16:49:16.097967Z","shell.execute_reply.started":"2024-12-01T16:49:15.644220Z","shell.execute_reply":"2024-12-01T16:49:16.096844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = ['feature_01']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T16:49:16.099288Z","iopub.execute_input":"2024-12-01T16:49:16.099595Z","iopub.status.idle":"2024-12-01T16:49:16.104694Z","shell.execute_reply.started":"2024-12-01T16:49:16.099559Z","shell.execute_reply":"2024-12-01T16:49:16.103532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"process_and_plot(df1, ['feature_01'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T16:57:52.795882Z","iopub.execute_input":"2024-12-01T16:57:52.796311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport plotly.express as px\nimport plotly.graph_objects as go\n\ndef analyze_features(data, feature1, feature2, max_lag=10):\n    \"\"\"\n    Analyzes the given features by plotting histograms, computing covariance,\n    and estimating time-lagged covariations.\n\n    Parameters:\n    - data: pandas DataFrame containing the features.\n    - feature1: First feature for covariance calculations.\n    - feature2: Second feature for covariance calculations.\n    - max_lag: Maximum time lag for time-lagged covariance.\n\n    Returns:\n    - lagged_covariances: List of lagged covariance values.\n    \"\"\"\n\n    # Plot Histograms for the Selected Features\n    def plot_histograms():\n        fig1 = px.histogram(data, x=feature1, title=f\"Histogram of {feature1}\", nbins=30, template=\"plotly_white\")\n        fig1.show()\n        fig2 = px.histogram(data, x=feature2, title=f\"Histogram of {feature2}\", nbins=30, template=\"plotly_white\")\n        fig2.show()\n\n    # Compute Covariance Between Two Features\n    def compute_covariance():\n        cov_matrix = data[[feature1, feature2]].cov()\n        covariance = cov_matrix.loc[feature1, feature2]\n        print(f\"Covariance between {feature1} and {feature2}: {covariance}\")\n        return covariance\n\n    # Estimate Time-Lagged Covariations\n    def compute_time_lagged_covariance():\n        lagged_covariances = []\n        for lag in range(1, max_lag + 1):\n            # Shift feature1 by the current lag\n            shifted_feature = data[feature1].shift(lag)\n            # Compute covariance between shifted feature1 and feature2\n            cov = np.cov(shifted_feature[lag:], data[feature2][lag:], ddof=0)[0, 1]\n            lagged_covariances.append(cov)\n        \n        # Plot the lagged covariances\n        fig = px.line(\n            x=np.arange(1, max_lag + 1),\n            y=lagged_covariances,\n            labels={'x': 'Lag', 'y': 'Lagged Covariance'},\n            title=f\"Time-Lagged Covariance Between {feature1} and {feature2}\",\n            template=\"plotly_white\"\n        )\n        fig.show()\n\n        return lagged_covariances\n\n    # Perform Analysis\n    plot_histograms()\n    covariance = compute_covariance()\n    lagged_covariances = compute_time_lagged_covariance()\n\n    return lagged_covariances\n\n# Example Usage\n# Sample DataFrame\nnp.random.seed(42)\ntrain = pd.DataFrame({\n    'feature_00': np.random.randn(100),\n    'feature_01': np.random.randn(100),\n    'feature_02': np.random.randn(100)\n})\n\n# Analyze features\nlagged_covs = analyze_features(train, 'feature_00', 'feature_01', max_lag=10)\nprint(\"Lagged Covariances:\", lagged_covs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:04:26.057061Z","iopub.execute_input":"2024-12-01T17:04:26.057495Z","iopub.status.idle":"2024-12-01T17:04:26.213469Z","shell.execute_reply.started":"2024-12-01T17:04:26.057458Z","shell.execute_reply":"2024-12-01T17:04:26.212263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyze_features(df1, 'feature_00', 'feature_01')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:04:55.348973Z","iopub.execute_input":"2024-12-01T17:04:55.349420Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport plotly.express as px\nimport plotly.graph_objects as go\n\ndef analyze_features(data, feature1, feature2, max_lag=10):\n    \"\"\"\n    Analyzes the given features by plotting histograms, computing covariance,\n    and estimating time-lagged covariations.\n\n    Parameters:\n    - data: pandas DataFrame containing the features.\n    - feature1: First feature for covariance calculations.\n    - feature2: Second feature for covariance calculations.\n    - max_lag: Maximum time lag for time-lagged covariance.\n\n    Returns:\n    - lagged_covariances: List of lagged covariance values.\n    \"\"\"\n\n    # Plot Histograms for the Selected Features\n    def plot_histograms():\n        fig1 = px.histogram(data, x=feature1, title=f\"Histogram of {feature1}\", nbins=30, template=\"plotly_white\")\n        fig1.show()\n        fig2 = px.histogram(data, x=feature2, title=f\"Histogram of {feature2}\", nbins=30, template=\"plotly_white\")\n        fig2.show()\n\n    # Compute Covariance Between Two Features\n    def compute_covariance():\n        cov_matrix = data[[feature1, feature2]].cov()\n        covariance = cov_matrix.loc[feature1, feature2]\n        print(f\"Covariance between {feature1} and {feature2}: {covariance}\")\n        return covariance\n\n    # Estimate Time-Lagged Covariations\n    def compute_time_lagged_covariance():\n        lagged_covariances = []\n        for lag in range(1, max_lag + 1):\n            # Shift feature1 by the current lag\n            shifted_feature = data[feature1].shift(lag)\n            # Compute covariance between shifted feature1 and feature2\n            cov = np.cov(shifted_feature[lag:], data[feature2][lag:], ddof=0)[0, 1]\n            lagged_covariances.append(cov)\n        \n        # Plot the lagged covariances\n        fig = px.line(\n            x=np.arange(1, max_lag + 1),\n            y=lagged_covariances,\n            labels={'x': 'Lag', 'y': 'Lagged Covariance'},\n            title=f\"Time-Lagged Covariance Between {feature1} and {feature2}\",\n            template=\"plotly_white\"\n        )\n        fig.show()\n\n        return lagged_covariances\n\n    # Perform Analysis\n    plot_histograms()\n    covariance = compute_covariance()\n    lagged_covariances = compute_time_lagged_covariance()\n\n    return lagged_covariances\n\n# Example Usage\n# Sample DataFrame\nnp.random.seed(42)\ntrain = pd.DataFrame({\n    'feature_00': np.random.randn(100),\n    'feature_01': np.random.randn(100),\n    'feature_02': np.random.randn(100)\n})\n\n# Analyze features\nlagged_covs = analyze_features(train, 'feature_00', 'feature_01', max_lag=10)\nprint(\"Lagged Covariances:\", lagged_covs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:00:58.363776Z","iopub.execute_input":"2024-12-01T17:00:58.364368Z","iopub.status.idle":"2024-12-01T17:00:58.517803Z","shell.execute_reply.started":"2024-12-01T17:00:58.364331Z","shell.execute_reply":"2024-12-01T17:00:58.516629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"analyze_features(df1, 'feature_00', 'feature_01')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:01:09.214011Z","iopub.execute_input":"2024-12-01T17:01:09.214442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom scipy.fft import fft, fftfreq\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import r2_score\nimport plotly.express as px\n\ndef analyze_timescales_and_predict(data, target, max_freq=None, top_peaks=5):\n    \"\"\"\n    Analyzes timescales using Fourier Transform and builds a predictive model\n    based on timescales, identifying the most important ones.\n\n    Parameters:\n    - data: pandas DataFrame containing time series data.\n    - target: Column name of the target variable for prediction.\n    - max_freq: Maximum frequency to consider (optional).\n    - top_peaks: Number of dominant frequencies to analyze.\n\n    Returns:\n    - model: Trained predictive model.\n    - r2: R^2 score on test set.\n    - important_timescales: List of important timescales identified.\n    \"\"\"\n\n    # Ensure the target variable exists\n    if target not in data.columns:\n        raise ValueError(f\"Target variable '{target}' not found in data.\")\n\n    # 1. Compute Fourier Transform for the target variable\n    y = data[target].fillna(0).values  # Replace NaNs with 0 for FFT\n    n = len(y)\n    yf = fft(y)\n    xf = fftfreq(n, d=1)[:n//2]  # Positive frequencies\n    amplitudes = 2.0 / n * np.abs(yf[:n//2])\n\n    # Limit to max frequency if specified\n    if max_freq:\n        mask = xf <= max_freq\n        xf = xf[mask]\n        amplitudes = amplitudes[mask]\n\n    # Identify the top peaks (relevant timescales)\n    peaks = np.argsort(amplitudes)[-top_peaks:]  # Indices of top peaks\n    relevant_frequencies = xf[peaks]\n    relevant_amplitudes = amplitudes[peaks]\n    relevant_timescales = 1 / relevant_frequencies  # Timescales in time units\n\n    # Plot the Fourier Transform\n    fig = px.line(x=xf, y=amplitudes, labels={'x': 'Frequency', 'y': 'Amplitude'},\n                  title=\"Fourier Transform - Amplitudes by Frequency\", template=\"plotly_white\")\n    for freq, amp in zip(relevant_frequencies, relevant_amplitudes):\n        fig.add_annotation(x=freq, y=amp, text=f\"Peak: {freq:.2f} Hz\", showarrow=True, arrowhead=2)\n    fig.show()\n\n    print(\"Relevant Timescales:\", relevant_timescales)\n\n    # 2. Create Features Based on Timescales\n    features = pd.DataFrame(index=data.index)\n    for freq, timescale in zip(relevant_frequencies, relevant_timescales):\n        # Generate sine and cosine components for each frequency\n        features[f'sin_{freq:.2f}'] = np.sin(2 * np.pi * freq * np.arange(len(data)))\n        features[f'cos_{freq:.2f}'] = np.cos(2 * np.pi * freq * np.arange(len(data)))\n\n    # Add lagged target as additional features\n    for lag in range(1, 5):\n        features[f'lag_{lag}'] = data[target].shift(lag)\n\n    # Drop NaN rows due to lagging\n    features = features.dropna()\n    target_values = data[target].iloc[features.index]\n\n    # 3. Train a Predictive Model\n    X_train, X_test, y_train, y_test = train_test_split(features, target_values, test_size=0.2, random_state=42)\n    model = LinearRegression()\n    model.fit(X_train, y_train)\n\n    # Evaluate the model\n    y_pred = model.predict(X_test)\n    r2 = r2_score(y_test, y_pred)\n    print(f\"R^2 Score on Test Set: {r2}\")\n\n    # Assess Timescale Importance\n    importance = pd.Series(model.coef_, index=features.columns).sort_values(ascending=False)\n    print(\"Feature Importance:\\n\", importance)\n\n    return model, r2, relevant_timescales, importance\n\n# Example Usage\nnp.random.seed(42)\ndata = pd.DataFrame({\n    'responder_6': np.sin(2 * np.pi * 0.1 * np.arange(100)) + 0.5 * np.random.randn(100)  # Sine wave + noise\n})\n\nmodel, r2, timescales, importance = analyze_timescales_and_predict(data, 'responder_6', max_freq=0.5, top_peaks=3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T17:00:54.033604Z","iopub.execute_input":"2024-12-01T17:00:54.033992Z","iopub.status.idle":"2024-12-01T17:00:54.986366Z","shell.execute_reply.started":"2024-12-01T17:00:54.033957Z","shell.execute_reply":"2024-12-01T17:00:54.985118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}