{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import data and libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib\nimport matplotlib.pyplot as plt  \nimport seaborn as sns\n%matplotlib inline\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-04T05:45:35.269790Z","iopub.execute_input":"2025-01-04T05:45:35.270326Z","iopub.status.idle":"2025-01-04T05:45:35.750238Z","shell.execute_reply.started":"2025-01-04T05:45:35.270287Z","shell.execute_reply":"2025-01-04T05:45:35.749249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df = pd.DataFrame()\nfor i in range(2):\n    partition = pd.read_parquet(f'/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id={i}/part-0.parquet')\n    sample_df = pd.concat([sample_df, partition])\n    del partition","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T05:45:36.141164Z","iopub.execute_input":"2025-01-04T05:45:36.141642Z","iopub.status.idle":"2025-01-04T05:45:45.074629Z","shell.execute_reply.started":"2025-01-04T05:45:36.141614Z","shell.execute_reply":"2025-01-04T05:45:45.073145Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Statistical Analysis","metadata":{}},{"cell_type":"code","source":"responder_col = [col for col in sample_df.columns if 'responder' in col]\nresponder_col","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T05:45:47.849174Z","iopub.execute_input":"2025-01-04T05:45:47.849593Z","iopub.status.idle":"2025-01-04T05:45:47.857392Z","shell.execute_reply.started":"2025-01-04T05:45:47.849566Z","shell.execute_reply":"2025-01-04T05:45:47.856072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T17:23:16.527728Z","iopub.execute_input":"2025-01-01T17:23:16.528153Z","iopub.status.idle":"2025-01-01T17:23:16.542416Z","shell.execute_reply.started":"2025-01-01T17:23:16.528117Z","shell.execute_reply":"2025-01-01T17:23:16.541160Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df.describe().T","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T17:24:43.761914Z","iopub.execute_input":"2025-01-01T17:24:43.762310Z","iopub.status.idle":"2025-01-01T17:25:01.543256Z","shell.execute_reply.started":"2025-01-01T17:24:43.762281Z","shell.execute_reply":"2025-01-01T17:25:01.541487Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Responder 6 by Symbol (Stock)","metadata":{}},{"cell_type":"markdown","source":"## Plot the Raw Time Series\n\nChart: Line plot\n- Plot stock prices (e.g., closing prices) over time to observe the overall trend.\n- If multivariate (e.g., multiple stocks), plot each stock individually.\n\nInterpretation: Look for long-term trends, seasonal patterns, and periods of volatility.","metadata":{}},{"cell_type":"code","source":"sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T17:59:06.621049Z","iopub.execute_input":"2025-01-01T17:59:06.621453Z","iopub.status.idle":"2025-01-01T17:59:06.722932Z","shell.execute_reply.started":"2025-01-01T17:59:06.621423Z","shell.execute_reply":"2025-01-01T17:59:06.721959Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Observations:**\nThere are some stocks (symbols) that does not have responder 6 values. \n\n**Plots:**\nExplore Responder 6 value over time for each available symbol","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(10, 2, figsize=(25, 50))\n\nfor i, symbol in enumerate(sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count().reset_index()['symbol_id']):\n    time = sample_df[sample_df['symbol_id']==symbol]['time_id']\n    target = sample_df[sample_df['symbol_id']==symbol]['responder_6']\n    ax = sns.lineplot(\n        data = sample_df,\n        x = time, \n        y = target, \n        color='black', \n        ax = axes[i // 2, i % 2])\n    ax.set_title(f'Responder 6 Over Time by Symbol {symbol}')\n    ax.set(xlabel='Time', ylabel='Return')\n    ax.grid(True)\n    ax.axhline(0, color='red', linestyle='-', linewidth=1.2)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T20:41:08.583355Z","iopub.execute_input":"2025-01-01T20:41:08.583746Z","iopub.status.idle":"2025-01-01T20:48:34.749102Z","shell.execute_reply.started":"2025-01-01T20:41:08.583712Z","shell.execute_reply":"2025-01-01T20:48:34.747163Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cumulative Returns","metadata":{}},{"cell_type":"markdown","source":"**Plot** - Explore `Responder 6` Cumulative Sum over time","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(10, 2, figsize=(25, 50))\n\nfor i, symbol in enumerate(sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count().reset_index()['symbol_id']):\n    time = sample_df[sample_df['symbol_id']==symbol]['date_id']\n    target = sample_df[sample_df['symbol_id']==symbol]['responder_6'].cumsum()\n    ax = sns.lineplot(\n        data = sample_df,\n        x = time, \n        y = target, \n        color='black', \n        ax = axes[i // 2, i % 2])\n    ax.set_title(f'Responder 6 Over Time by Symbol {symbol}')\n    ax.set(xlabel='Time', ylabel='Cumulative Return')\n    ax.grid(True)\n    ax.axhline(0, color='red', linestyle='-', linewidth=1.2)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T20:48:34.750746Z","iopub.execute_input":"2025-01-01T20:48:34.751406Z","iopub.status.idle":"2025-01-01T20:51:33.954705Z","shell.execute_reply.started":"2025-01-01T20:48:34.751350Z","shell.execute_reply":"2025-01-01T20:51:33.953448Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Percentage Change in Daily Returns","metadata":{}},{"cell_type":"markdown","source":"Metric: Daily Returns\n- Calculate percentage returns: pct_change()\n- Chart: Histogram + Time Series plot of returns.\n  \nInterpretation: Identify periods of extreme returns (e.g., crashes, spikes). Look for clustering of volatility (GARCH-like behavior).","metadata":{}},{"cell_type":"code","source":"sample_df['pct_change'] = sample_df.groupby(['time_id', 'symbol_id'])['responder_6'].pct_change()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T22:22:38.751173Z","iopub.execute_input":"2025-01-02T22:22:38.751597Z","iopub.status.idle":"2025-01-02T22:22:41.097615Z","shell.execute_reply.started":"2025-01-02T22:22:38.751564Z","shell.execute_reply":"2025-01-02T22:22:41.096616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(10, 2, figsize=(25, 100)) \naxes = axes.flatten()  \n\nfor i, symbol in enumerate(sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count().reset_index()['symbol_id']):\n    ax = axes[i] \n    values, bins, bars = ax.hist(sample_df[(sample_df['symbol_id']==0) & (sample_df['pct_change'].between(-50,50))]['pct_change'], alpha=0.7, bins = 30)\n    ax.set_xlabel('Percentage Change (%)Daily Returns')\n    ax.set_title(f'Distribution of Daily Returns of Symbol {symbol}', weight='bold')\n    ax.bar_label(bars)\n    ax.grid(visible=True, color='gray', linewidth=0.7)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T22:30:51.375321Z","iopub.execute_input":"2025-01-02T22:30:51.375804Z","iopub.status.idle":"2025-01-02T22:31:08.403411Z","shell.execute_reply.started":"2025-01-02T22:30:51.375767Z","shell.execute_reply":"2025-01-02T22:31:08.401920Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Outlier Detection","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(10, 2, figsize=(25, 90))\naxes = axes.flatten()  # Flatten the axes for easier indexing\n\nfor i, symbol in enumerate(sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count().reset_index()['symbol_id']):\n    # Filter data for the current symbol\n    symbol_data = sample_df[sample_df['symbol_id'] == symbol]['responder_6'].dropna()\n\n    # Calculate IQR and extreme outlier bounds\n    q1 = symbol_data.quantile(0.25)\n    q3 = symbol_data.quantile(0.75)\n    iqr = q3 - q1\n    lower_bound = q1 - 3 * iqr  # Extreme lower outlier bound\n    upper_bound = q3 + 3 * iqr  # Extreme upper outlier bound\n\n    # Plot boxplot\n    ax = axes[i]\n    sns.boxplot(y=symbol_data, color='skyblue', ax=ax)  # Use x= for Series data\n    ax.set_title(f'Distribution of Responder 6 by Symbol {symbol}', weight='bold')\n\n    # Add red lines for extreme outlier boundaries\n    ax.axhline(lower_bound, color='red', linestyle='--', linewidth=1, label='Extreme Outlier Lower Bound')\n    ax.axhline(upper_bound, color='red', linestyle='--', linewidth=1, label='Extreme Outlier Upper Bound')\n\n    # Add legend only once per plot\n    ax.legend()\n\n    # Enable grid for better readability\n    ax.grid(True)\n\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T22:46:05.075632Z","iopub.execute_input":"2025-01-02T22:46:05.076021Z","iopub.status.idle":"2025-01-02T22:46:14.724319Z","shell.execute_reply.started":"2025-01-02T22:46:05.075995Z","shell.execute_reply":"2025-01-02T22:46:14.722845Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Seasonal-Trend decomposition using LOESS (STL)","metadata":{}},{"cell_type":"code","source":"from statsmodels.tsa.seasonal import STL\nfor i, symbol in enumerate(sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count().reset_index()['symbol_id']):\n    stl = STL(sample_df[sample_df['symbol_id']==symbol]['responder_6'], period=12, seasonal=13)\n    result = stl.fit()\n    fig, (ax1, ax2, ax3, ax4) = plt.subplots(4, 1, figsize=(10, 8), sharex=True)\n    ax1.plot(sample_df[sample_df['symbol_id']==symbol]['responder_6'])\n    ax1.set_title(f'Original Responder 6 for Symbol {symbol}')\n    ax2.plot(result.trend)\n    ax2.set_title(f'Trend Component for Symbol {symbol}')\n    ax3.plot(result.seasonal)\n    ax3.set_title(f'Seasonal Component for Symbol {symbol}')\n    ax4.plot(result.resid)\n    ax4.set_title(f'Residual (Noise) Component for Symbol {symbol}')\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T21:51:40.930870Z","iopub.execute_input":"2025-01-02T21:51:40.931235Z","iopub.status.idle":"2025-01-02T21:53:25.797600Z","shell.execute_reply.started":"2025-01-02T21:51:40.931208Z","shell.execute_reply":"2025-01-02T21:53:25.796396Z"},"_kg_hide-output":false,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Decompose the Time Series\nChart: Seasonal Decomposition (Trend + Seasonality + Residuals)\n- Decompose the series into components using statsmodels or seasonal_decompose.\n\nInterpretation: Separate the overall trend, seasonality (e.g., monthly cycles), and noise.","metadata":{}},{"cell_type":"code","source":"from statsmodels.tsa.seasonal import seasonal_decompose\nfor i, symbol in enumerate(sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count().reset_index()['symbol_id']):\n    responder_6_decomp = seasonal_decompose(sample_df[sample_df['symbol_id']==symbol]['responder_6'], model='additive', period=251)\n    fig = responder_6_decomp.plot()  # Plot the decomposition\n    fig.suptitle(f'Time Series Decomposition for Symbol {symbol}', fontsize=12)  \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T21:32:59.287799Z","iopub.execute_input":"2025-01-01T21:32:59.288187Z","iopub.status.idle":"2025-01-01T21:33:37.383746Z","shell.execute_reply.started":"2025-01-01T21:32:59.288158Z","shell.execute_reply":"2025-01-01T21:33:37.381829Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Moving Average\nChart: Rolling Mean \n- Plot moving averages (e.g., 20-day, 50-day) to smooth short-term fluctuations.\n\nInterpretation: Use rolling mean to spot trends","metadata":{}},{"cell_type":"code","source":"stat_df = sample_df.copy()\nfig, axes = plt.subplots(10, 2, figsize=(25, 50)) \naxes = axes.flatten() \n\nfor i, symbol in enumerate(sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count().reset_index()['symbol_id']):\n    symbol_data = stat_df[stat_df['symbol_id'] == symbol]\n    ax = axes[i]\n    ax.plot(symbol_data['responder_6'].rolling(window=20).mean(), alpha = 0.5, label=f'20-day Moving Average for Symbol {symbol}')\n    ax.plot(symbol_data['responder_6'].rolling(window=100).mean(), alpha = 0.9, label=f'100-day Moving Average for Symbol {symbol}')\n    ax.set_title(f'Moving Average of Responder 6 for Symbol {symbol}', weight='bold')\n    ax.grid(visible=True, color='gray', linewidth=0.7)\n    ax.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T21:25:24.889024Z","iopub.execute_input":"2025-01-02T21:25:24.889431Z","iopub.status.idle":"2025-01-02T21:26:08.802894Z","shell.execute_reply.started":"2025-01-02T21:25:24.889403Z","shell.execute_reply":"2025-01-02T21:26:08.800940Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Volatility Analysis\n\nMetric: Rolling Standard Deviation or ATR (Average True Range)\n- Use to gauge periods of high vs. low volatility.\n\nChart: Line plot to spot volatility","metadata":{}},{"cell_type":"code","source":"stat_df = sample_df.copy()\nfig, axes = plt.subplots(10, 2, figsize=(25, 50)) \naxes = axes.flatten() \n\nfor i, symbol in enumerate(sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count().reset_index()['symbol_id']):\n    symbol_data = stat_df[stat_df['symbol_id'] == symbol]\n    ax = axes[i]\n    ax.plot(symbol_data['responder_6'].rolling(window=20).std(), alpha = 0.5, label=f'20-day Moving Average for Symbol {symbol}')\n    ax.plot(symbol_data['responder_6'].rolling(window=100).std(), alpha = 0.9, label=f'100-day Moving Average for Symbol {symbol}')\n    ax.set_title(f'Moving Standard Deviation of Responder 6 for Symbol {symbol}', weight='bold')\n    ax.grid(visible=True, color='gray', linewidth=0.7)\n    ax.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T21:26:08.805083Z","iopub.execute_input":"2025-01-02T21:26:08.805468Z","iopub.status.idle":"2025-01-02T21:26:38.445092Z","shell.execute_reply.started":"2025-01-02T21:26:08.805434Z","shell.execute_reply":"2025-01-02T21:26:38.443649Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Autocorrelation Analysis\nChart: Autocorrelation Function (ACF) and Partial Autocorrelation Function (PACF)\n- Examine the correlation of a series with its past values.\n- Interpretation: Identify whether the series has short-term dependencies (lags).\n\nSignificant lags in ACF indicate patterns or seasonality.\n\n[More on how to intepret the plots](https://medium.com/@kis.andras.nandor/understanding-autocorrelation-and-partial-autocorrelation-functions-acf-and-pacf-2998e7e1bcb5)","metadata":{}},{"cell_type":"code","source":"from statsmodels.graphics.tsaplots import plot_acf, plot_pacf\n\nfor i, symbol in enumerate(sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count().reset_index()['symbol_id']):\n    fig = plot_acf(sample_df[sample_df['symbol_id']==symbol]['responder_6'].dropna(), lags=100)\n    fig.suptitle(f'Autocorrelation Analysis for Symbol {symbol}', fontsize=12)  \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T05:46:03.040647Z","iopub.execute_input":"2025-01-04T05:46:03.041055Z","iopub.status.idle":"2025-01-04T05:50:05.245128Z","shell.execute_reply.started":"2025-01-04T05:46:03.041026Z","shell.execute_reply":"2025-01-04T05:50:05.243925Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from statsmodels.graphics.tsaplots import plot_acf, plot_pacf\n\nfor i, symbol in enumerate(sample_df[['responder_6', 'symbol_id']].groupby('symbol_id').count().reset_index()['symbol_id']):\n    fig = plot_pacf(sample_df[sample_df['symbol_id']==symbol]['responder_6'].dropna())\n    fig.suptitle(f'Partial Autocorrelation Analysis for Symbol {symbol}', fontsize=12)  \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-04T05:50:05.246663Z","iopub.execute_input":"2025-01-04T05:50:05.247104Z","iopub.status.idle":"2025-01-04T05:50:27.551199Z","shell.execute_reply.started":"2025-01-04T05:50:05.247067Z","shell.execute_reply":"2025-01-04T05:50:27.550018Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Responder 6 vs. Other Responders","metadata":{}},{"cell_type":"markdown","source":"## Responders Over Time\n**Plot:** Compare `Responder 6` with other responders, using only `symbol_id = 0`","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(9, 1, figsize=(25, 90)) \naxes = axes.flatten()  \n\nfor i, responder in enumerate(responder_col): \n    ax = axes[i] \n    \n    if responder != 'responder_6':  \n        c = 'red'\n        lw = 2.5\n        ax.plot(\n            (sample_df[sample_df.symbol_id == 0].groupby(['date_id'])['responder_6'].mean()).cumsum(), \n            linewidth=lw, color=c, label='Responder 6'\n        )\n        \n        # Plot the other responder\n        lw = 1\n        ax.plot(\n            (sample_df[sample_df.symbol_id == 0].groupby(['date_id'])[responder].mean()).cumsum(), \n            linewidth=lw, label=responder\n        )\n        \n        # Subplot settings\n        ax.set_xlabel('Trade days')\n        ax.set_ylabel('Cumulative response')\n        ax.set_title(f'Response time series over trade days  \\n Responder 6 (red) and {responder}', weight='bold')\n        ax.grid(visible=True, color='gray', linewidth=0.7)\n        ax.axhline(0, color='red', linestyle='-', linewidth=1)\n        ax.legend()\n    else:\n        ax.plot(\n        (sample_df[sample_df.symbol_id == 0].groupby(['date_id'])['responder_6'].mean()).cumsum(), \n        linewidth=lw, color=c, label='Responder 6'\n        )\n        ax.set_xlabel('Trade days')\n        ax.set_ylabel('Cumulative response')\n        ax.set_title('Response time series over trade days  \\n Responder 6 (red)', weight='bold')\n        ax.grid(visible=True, color='gray', linewidth=0.7)\n        ax.axhline(0, color='red', linestyle='-', linewidth=1)\n        ax.legend()\n        \n        \n# Adjust layout for better spacing\nplt.tight_layout()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-01T18:55:25.675939Z","iopub.execute_input":"2025-01-01T18:55:25.676343Z","iopub.status.idle":"2025-01-01T18:55:33.539050Z","shell.execute_reply.started":"2025-01-01T18:55:25.676309Z","shell.execute_reply":"2025-01-01T18:55:33.537848Z"},"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Cross-Correlation\nhttps://www.kaggle.com/code/allegich/jane-street-time-series-analysis-eda-ensemble#-TIME-SERIES-ANALYSIS-AND-EDA","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(6, 6))\npath = \"/kaggle/input/jane-street-real-time-market-data-forecasting\"\nresponders = pd.read_csv(f\"{path}/responders.csv\")\nmatrix = responders[[ f\"tag_{no}\" for no in range(0,5,1) ] ].T.corr()\nsns.heatmap(matrix, square=True, cmap=\"coolwarm\", alpha =0.9, vmin=-1, vmax=1, center= 0, linewidths=0.5, \n            linecolor='white', annot=True, fmt='.2f')\nplt.xlabel(\"Responder_0 - Responder_8\")\nplt.ylabel(\"Responder_0 - Responder_8\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T22:54:15.710810Z","iopub.execute_input":"2025-01-02T22:54:15.711239Z","iopub.status.idle":"2025-01-02T22:54:16.203575Z","shell.execute_reply.started":"2025-01-02T22:54:15.711205Z","shell.execute_reply":"2025-01-02T22:54:16.202565Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Distribution by Responder\n**Plots:** Returns and cumulative daily returns, and disribution of returns for all responders in regards to symbol_id = 0","metadata":{}},{"cell_type":"code","source":"sample_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T19:42:52.736911Z","iopub.execute_input":"2025-01-03T19:42:52.737285Z","iopub.status.idle":"2025-01-03T19:42:53.335718Z","shell.execute_reply.started":"2025-01-03T19:42:52.737253Z","shell.execute_reply":"2025-01-03T19:42:53.334759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_df\ngridColor = 'lightgrey'\nrow = len(responder_col)\nj = 0\n\nfig, axs = plt.subplots(figsize=(18, 4*row))\nfor i in range(1, 3 * len(responder_col) + 1, 3):\n    xx=sample_df[(sample_df.symbol_id==0)]['date_id']\n    yy=sample_df[ (sample_df.symbol_id==0)][f'responder_{j}']\n    c='black'\n    if j == 6: c='red'\n        \n    ax1 = plt.subplot(9, 3, i)\n    ax1.plot(xx,yy.cumsum(), color = c, linewidth =0.8 )\n    plt.axhline(0, color='blue', linestyle='-', linewidth=0.9)\n    plt.grid(color =gridColor )\n    \n    ax2 = plt.subplot(9, 3, i+1)\n    ax2.plot(xx,yy, color = c, linewidth =0.05)\n    plt.axhline(0, color='blue', linestyle='-', linewidth=1.2)\n    ax2.set_title(f\"responder_{j}\", fontsize = 14)\n    plt.grid(color = gridColor)\n    \n    ax3 = plt.subplot(9, 3, i+2)\n    b=1000\n    ax3.hist(yy, bins=b, color = c,density=True, histtype=\"step\" )\n    ax3.hist(yy, bins=b, color = 'lightgrey',density=True)\n    plt.grid(color = gridColor)\n    ax3.set_ylim([0, 3.5])\n    ax3.set_xlim([-2.5, 2.5])\n    \n    j = j + 1\n    \nfig.patch.set_linewidth(3)\nfig.patch.set_edgecolor('#000000')\nfig.patch.set_facecolor('#eeeeee') \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-03T19:49:10.126442Z","iopub.execute_input":"2025-01-03T19:49:10.126874Z","iopub.status.idle":"2025-01-03T19:49:30.657158Z","shell.execute_reply.started":"2025-01-03T19:49:10.126838Z","shell.execute_reply":"2025-01-03T19:49:30.655999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}