{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:24:39.113026Z","iopub.execute_input":"2025-07-27T02:24:39.113272Z","iopub.status.idle":"2025-07-27T02:24:39.497219Z","shell.execute_reply.started":"2025-07-27T02:24:39.113249Z","shell.execute_reply":"2025-07-27T02:24:39.496217Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# The plots and there uses are for goals\n    1. How our time series data is better represented.\n    2. What's the structure of data.\n    3. What's the pattern of data","metadata":{}},{"cell_type":"code","source":"import time\n# load training data\nt1 = time.time()\ndf_train = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\nt2 = time.time()\nprint('Elapsed time [s]:', np.round(t2-t1, 2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:24:39.499049Z","iopub.execute_input":"2025-07-27T02:24:39.499432Z","iopub.status.idle":"2025-07-27T02:25:05.677153Z","shell.execute_reply.started":"2025-07-27T02:24:39.499407Z","shell.execute_reply":"2025-07-27T02:25:05.676174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# preview\ndf_train.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:05.678514Z","iopub.execute_input":"2025-07-27T02:25:05.678877Z","iopub.status.idle":"2025-07-27T02:25:05.718202Z","shell.execute_reply.started":"2025-07-27T02:25:05.678841Z","shell.execute_reply":"2025-07-27T02:25:05.717284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.core.interactiveshell import InteractiveShell\nInteractiveShell.ast_node_interactivity = \"all\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:05.719214Z","iopub.execute_input":"2025-07-27T02:25:05.719564Z","iopub.status.idle":"2025-07-27T02:25:05.724332Z","shell.execute_reply.started":"2025-07-27T02:25:05.719529Z","shell.execute_reply":"2025-07-27T02:25:05.723368Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# train.parquet\n## The training dataset containing all historical market data along with the corresponding labels.\n\n### timestamp: The timestamp index representing the minute associated with each row.\n### bid_qty: The total quantity buyers are willing to purchase at the best (highest) bid price at the given timestamp.\n### ask_qty: The total quantity sellers are offering to sell at the best (lowest) ask price at the given timestamp.\n### buy_qty: The total trading quantity executed at the best ask price during the given minute.\n### sell_qty: The total trading quantity executed at the best bid price during the given minute.\n### volume: The total traded volume during the minute.\n### X_{1,...,780}: A set of anonymized market features derived from proprietary data sources.\n### label: The target variable representing the anonymized market price movement to be predicted.\n\n# structure details\n# df_train.info(verbose=True, show_counts=True)\n\n# main features\nfeatures_main = ['bid_qty', 'ask_qty', 'buy_qty', 'sell_qty', 'volume']\n\n# anonymized features\nfeatures_x = ['X' + str(i) for i in range(1,890+1)]\n\n# target\ntarget = 'label'","metadata":{"trusted":true,"_kg_hide-input":false,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:05.725358Z","iopub.execute_input":"2025-07-27T02:25:05.725679Z","iopub.status.idle":"2025-07-27T02:25:05.745237Z","shell.execute_reply.started":"2025-07-27T02:25:05.725650Z","shell.execute_reply":"2025-07-27T02:25:05.744214Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Supress Warning","metadata":{}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:05.746190Z","iopub.execute_input":"2025-07-27T02:25:05.746504Z","iopub.status.idle":"2025-07-27T02:25:05.769762Z","shell.execute_reply.started":"2025-07-27T02:25:05.746477Z","shell.execute_reply":"2025-07-27T02:25:05.768686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print (\"Rows     : \" ,df_train.shape[0])\nprint (\"Columns  : \" ,df_train.shape[1])\nprint (\"\\nMissing values :  \", df_train.isnull().any())\nprint (\"\\nUnique values :  \\n\",df_train.nunique())\n     \n\ndf_train_non_indexed=df_train.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:05.773399Z","iopub.execute_input":"2025-07-27T02:25:05.773720Z","iopub.status.idle":"2025-07-27T02:25:33.461954Z","shell.execute_reply.started":"2025-07-27T02:25:05.773693Z","shell.execute_reply":"2025-07-27T02:25:33.460780Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.reset_index(drop=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:33.462866Z","iopub.execute_input":"2025-07-27T02:25:33.463173Z","iopub.status.idle":"2025-07-27T02:25:35.480838Z","shell.execute_reply.started":"2025-07-27T02:25:33.463146Z","shell.execute_reply":"2025-07-27T02:25:35.479015Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from datetime import datetime","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:35.481908Z","iopub.execute_input":"2025-07-27T02:25:35.482238Z","iopub.status.idle":"2025-07-27T02:25:35.486702Z","shell.execute_reply.started":"2025-07-27T02:25:35.482210Z","shell.execute_reply":"2025-07-27T02:25:35.485847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:35.487847Z","iopub.execute_input":"2025-07-27T02:25:35.488151Z","iopub.status.idle":"2025-07-27T02:25:35.519782Z","shell.execute_reply.started":"2025-07-27T02:25:35.488119Z","shell.execute_reply":"2025-07-27T02:25:35.519006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_data=df_train['label']\nlabel_data.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:35.520746Z","iopub.execute_input":"2025-07-27T02:25:35.521168Z","iopub.status.idle":"2025-07-27T02:25:35.531625Z","shell.execute_reply.started":"2025-07-27T02:25:35.521084Z","shell.execute_reply":"2025-07-27T02:25:35.530024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\n\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:35.532580Z","iopub.execute_input":"2025-07-27T02:25:35.532951Z","iopub.status.idle":"2025-07-27T02:25:36.616513Z","shell.execute_reply.started":"2025-07-27T02:25:35.532919Z","shell.execute_reply":"2025-07-27T02:25:36.615543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_data.plot(grid=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:36.617572Z","iopub.execute_input":"2025-07-27T02:25:36.618050Z","iopub.status.idle":"2025-07-27T02:25:38.384816Z","shell.execute_reply.started":"2025-07-27T02:25:36.618021Z","shell.execute_reply":"2025-07-27T02:25:38.383869Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Considering Month by month","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\n\n# # Ensure the index is a DatetimeIndex\n# df_train.index = pd.to_datetime(df_train.index)\n\n# # Loop from March to December 2023\n# for month in range(3, 13):  # 3 to 12 inclusive\n#     month_str = f'2023-{month:02d}'  # e.g., '2023-03', '2023-04', etc.\n#     df_month = df_train.loc[month_str]\n\n#     if 'label' in df_month.columns:\n#         df_label_month = df_month['label']\n\n#         plt.figure(figsize=(10, 4))\n#         df_label_month.plot(grid=True, title=f'Label Plot - {month_str}');\n#         plt.xlabel('Date')\n#         plt.ylabel('Label')\n#         plt.tight_layout()\n#         plt.show()\n#     else:\n#         print(f\"'label' column not found for {month_str}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:38.385854Z","iopub.execute_input":"2025-07-27T02:25:38.386643Z","iopub.status.idle":"2025-07-27T02:25:38.391029Z","shell.execute_reply.started":"2025-07-27T02:25:38.386612Z","shell.execute_reply":"2025-07-27T02:25:38.390236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n# import pandas as pd\n\n# # Ensure datetime index\n# df_train.index = pd.to_datetime(df_train.index)\n\n# # Define simplified ranges\n# ranges = {\n#     '[-2.5, 2.5]': lambda x: (x >= -2.5) & (x <= 2.5),\n#     '(-5, -2.5) ∪ (2.5, 5]': lambda x: ((x > 2.5) & (x <= 5)) | ((x > -5) & (x < -2.5)),\n#     'Extreme (< -5 or > 5)': lambda x: (x <= -5) | (x > 5)\n# }\n\n# # Loop through March to December 2023\n# # for month in range(3, 13):\n#     month_str = f'2023-{month:02d}'\n#     df_month = df_train.loc[month_str]\n\n#     if 'label' not in df_month.columns:\n#         print(f\"'label' column not found in {month_str}\")\n#         continue\n\n#     df_label = df_month['label']\n\n#     # --- Plot Setup ---\n#     plt.figure(figsize=(14, 5))\n\n#     # Line plot\n#     plt.subplot(1, 2, 1)\n#     df_label.plot(grid=True, title=f'Label Time Series - {month_str}')\n#     plt.xlabel('Date')\n#     plt.ylabel('Label')\n\n#     # Box plot\n#     plt.subplot(1, 2, 2)\n#     box_data = []\n\n#     for label_range, condition in ranges.items():\n#         filtered = df_label[condition(df_label)]\n#         box_data.append(filtered)\n\n#     plt.boxplot(box_data, labels=ranges.keys(), showfliers=False)\n#     plt.title(f'Label Box Plot by Group - {month_str}')\n#     plt.ylabel('Label')\n\n#     plt.tight_layout()\n#     plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:38.392078Z","iopub.execute_input":"2025-07-27T02:25:38.392832Z","iopub.status.idle":"2025-07-27T02:25:38.416013Z","shell.execute_reply.started":"2025-07-27T02:25:38.392780Z","shell.execute_reply":"2025-07-27T02:25:38.414642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df=df_train.reset_index()\ndf.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:38.417016Z","iopub.execute_input":"2025-07-27T02:25:38.417359Z","iopub.status.idle":"2025-07-27T02:25:40.253535Z","shell.execute_reply.started":"2025-07-27T02:25:38.417333Z","shell.execute_reply":"2025-07-27T02:25:40.252559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# You processed it and saved the needed output, now delete 'df_train' as it is no longer needed\ndel df_train\n\n# Import garbage collector and free memory\nimport gc\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:40.254526Z","iopub.execute_input":"2025-07-27T02:25:40.254783Z","iopub.status.idle":"2025-07-27T02:25:40.447203Z","shell.execute_reply.started":"2025-07-27T02:25:40.254762Z","shell.execute_reply":"2025-07-27T02:25:40.446425Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# More convenient way of understanding target variable","metadata":{}},{"cell_type":"markdown","source":"## Label slider","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\nfig = px.line(df, x='index', y='label', title='label with Slider')\n\nfig.update_xaxes(rangeslider_visible=True)\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:40.448067Z","iopub.execute_input":"2025-07-27T02:25:40.448446Z","iopub.status.idle":"2025-07-27T02:25:51.731489Z","shell.execute_reply.started":"2025-07-27T02:25:40.448411Z","shell.execute_reply":"2025-07-27T02:25:51.729899Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Volume Slider","metadata":{}},{"cell_type":"code","source":"# import plotly.express as px\n# fig = px.line(df, x='index', y='volume', title='volume with Slider')\n\n# fig.update_xaxes(rangeslider_visible=True)\n# fig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:51.733069Z","iopub.execute_input":"2025-07-27T02:25:51.733622Z","iopub.status.idle":"2025-07-27T02:25:51.739238Z","shell.execute_reply.started":"2025-07-27T02:25:51.733564Z","shell.execute_reply":"2025-07-27T02:25:51.738006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:51.740383Z","iopub.execute_input":"2025-07-27T02:25:51.740731Z","iopub.status.idle":"2025-07-27T02:25:51.765931Z","shell.execute_reply.started":"2025-07-27T02:25:51.740699Z","shell.execute_reply":"2025-07-27T02:25:51.764879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fig = px.line(df, x='index', y='bid_qty', title='bid qty with Slider')\n\n# fig.update_xaxes(\n#     rangeslider_visible=True,\n#     rangeselector=dict(\n#         buttons=list([\n#             dict(count=1, label=\"1m\", step=\"month\", stepmode=\"backward\"),\n#             dict(count=2, label=\"2m\", step=\"month\", stepmode=\"backward\"),\n#             dict(count=3, label=\"3m\", step=\"month\", stepmode=\"backward\"),\n#             dict(step=\"all\")\n#         ])\n#     )\n# )\n# fig.show()\n     ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:51.767494Z","iopub.execute_input":"2025-07-27T02:25:51.767864Z","iopub.status.idle":"2025-07-27T02:25:51.793757Z","shell.execute_reply.started":"2025-07-27T02:25:51.767832Z","shell.execute_reply":"2025-07-27T02:25:51.778775Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_main","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:51.798469Z","iopub.execute_input":"2025-07-27T02:25:51.798912Z","iopub.status.idle":"2025-07-27T02:25:51.804876Z","shell.execute_reply.started":"2025-07-27T02:25:51.798887Z","shell.execute_reply":"2025-07-27T02:25:51.803721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Ensure datetime index\n# df.index = pd.to_datetime(df.index)\n\n# # Slice date range and select main features\n# df_filtered = df['2023':'2024'][features_main].copy()\n# df_filtered['month'] = df_filtered.index.to_period('M')  # or use .month for numeric\n\n# # Group by month and get descriptive stats\n# monthly_stats = df_filtered.groupby('month').describe()\n\n# print(monthly_stats)\n\n# df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:51.805591Z","iopub.execute_input":"2025-07-27T02:25:51.805903Z","iopub.status.idle":"2025-07-27T02:25:51.819929Z","shell.execute_reply.started":"2025-07-27T02:25:51.805879Z","shell.execute_reply":"2025-07-27T02:25:51.818768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:51.820900Z","iopub.execute_input":"2025-07-27T02:25:51.821204Z","iopub.status.idle":"2025-07-27T02:25:53.657947Z","shell.execute_reply.started":"2025-07-27T02:25:51.821180Z","shell.execute_reply":"2025-07-27T02:25:53.657125Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reducing Noise by rolling window:\n\nShow the very first row (00:00:00) as it is.\n\nFor every row where minute % 5 == 0 (except the first, which is shown raw), display the sum over the previous five rows (including that row).","metadata":{}},{"cell_type":"code","source":"# Rolling sum with a 5-row window\nwindow = 5\nrolling_sum = df.rolling(window=window, min_periods=window).sum()\n\n# Identify rows to keep: first row, plus every minute % 5 == 0 after the first row\nto_show = (df.index == df.index[0]) | ((df.index.minute % 5 == 0) & (df.index != df.index[0]))\n\n# Combine: for the first row, show original; for 5-minute marks, show rolling sum\nresult = df.copy()\nresult[to_show] = rolling_sum[to_show]\nresult.iloc[0] = df.iloc[0]  # Ensure first row is as-is\n\n# print(result[to_show])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:25:53.658783Z","iopub.execute_input":"2025-07-27T02:25:53.659063Z","iopub.status.idle":"2025-07-27T02:26:16.387883Z","shell.execute_reply.started":"2025-07-27T02:25:53.659035Z","shell.execute_reply":"2025-07-27T02:26:16.386864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:16.388563Z","iopub.execute_input":"2025-07-27T02:26:16.388849Z","iopub.status.idle":"2025-07-27T02:26:16.395871Z","shell.execute_reply.started":"2025-07-27T02:26:16.388818Z","shell.execute_reply":"2025-07-27T02:26:16.394893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# You processed it and saved the needed output, now delete 'large_df' as it is no longer needed\ndel df\n\n# Import garbage collector and free memory\nimport gc\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:16.396727Z","iopub.execute_input":"2025-07-27T02:26:16.397118Z","iopub.status.idle":"2025-07-27T02:26:16.651077Z","shell.execute_reply.started":"2025-07-27T02:26:16.397081Z","shell.execute_reply":"2025-07-27T02:26:16.650157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result[to_show].tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:16.652214Z","iopub.execute_input":"2025-07-27T02:26:16.652489Z","iopub.status.idle":"2025-07-27T02:26:17.146850Z","shell.execute_reply.started":"2025-07-27T02:26:16.652468Z","shell.execute_reply":"2025-07-27T02:26:17.146128Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=result[to_show]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:17.147739Z","iopub.execute_input":"2025-07-27T02:26:17.148022Z","iopub.status.idle":"2025-07-27T02:26:17.625360Z","shell.execute_reply.started":"2025-07-27T02:26:17.147992Z","shell.execute_reply":"2025-07-27T02:26:17.624338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_main","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:17.626478Z","iopub.execute_input":"2025-07-27T02:26:17.627092Z","iopub.status.idle":"2025-07-27T02:26:17.632960Z","shell.execute_reply.started":"2025-07-27T02:26:17.627060Z","shell.execute_reply":"2025-07-27T02:26:17.632198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import matplotlib.pyplot as plt\n\n# # Replace 'features_main' with your actual feature list\n# # Example: features_main = ['X1', 'X2', 'X3', ..., 'Xn']\n# # And train is your DataFrame\n\n# axes = train[features_main].hist(\n#     bins=20, \n#     figsize=(15, 15)\n# )\n\n# for ax in axes.flatten():\n#     ax.set_xlim(0, 1000)  # Set x-axis from 0 to 1\n#     ax.set_xlabel(\"Value (0-100)\")\n\n# plt.tight_layout()\n# plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:17.633969Z","iopub.execute_input":"2025-07-27T02:26:17.634364Z","iopub.status.idle":"2025-07-27T02:26:17.651137Z","shell.execute_reply.started":"2025-07-27T02:26:17.634334Z","shell.execute_reply":"2025-07-27T02:26:17.650235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[['label']].plot(kind='density')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:17.652208Z","iopub.execute_input":"2025-07-27T02:26:17.652475Z","iopub.status.idle":"2025-07-27T02:26:19.912738Z","shell.execute_reply.started":"2025-07-27T02:26:17.652453Z","shell.execute_reply":"2025-07-27T02:26:19.911786Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.plotting.lag_plot(train['label'],lag=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:19.913749Z","iopub.execute_input":"2025-07-27T02:26:19.913996Z","iopub.status.idle":"2025-07-27T02:26:20.554496Z","shell.execute_reply.started":"2025-07-27T02:26:19.913978Z","shell.execute_reply":"2025-07-27T02:26:20.553583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# You processed it and saved the needed output, now delete 'large_df' as it is no longer needed\ndel train\n\n# Import garbage collector and free memory\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:20.555449Z","iopub.execute_input":"2025-07-27T02:26:20.555756Z","iopub.status.idle":"2025-07-27T02:26:20.775597Z","shell.execute_reply.started":"2025-07-27T02:26:20.555733Z","shell.execute_reply":"2025-07-27T02:26:20.774734Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Autocorrelation\n\n**Autocorrelation** measures how a variable is correlated with itself over time. It is often used in time series analysis to check if previous values in a series predict future values.\n\n- Checks correlation of values with their own past values (lags).\n- Positive autocorrelation means high values tend to follow high values.\n- Useful for detecting trends or seasonality.\n\n\n## After one trading day co-relation , 7 hours=420 minutes in day, our data 420/5=84 rows\n### 1 day after lag as Traing hour in a day are 9.30-3.30, Total 7 hours","metadata":{}},{"cell_type":"code","source":"pd.plotting.lag_plot(result[to_show]['label'],lag=85)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:20.776632Z","iopub.execute_input":"2025-07-27T02:26:20.776937Z","iopub.status.idle":"2025-07-27T02:26:21.889937Z","shell.execute_reply.started":"2025-07-27T02:26:20.776905Z","shell.execute_reply":"2025-07-27T02:26:21.888957Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**No corelation at all, as data is donut star shaped**","metadata":{}},{"cell_type":"markdown","source":"## Pearson Correlation\n\nPearson correlation coefficient measures the **linear** relationship between two continuous variables.\n\n- Value ranges from -1 (perfect negative correlation) to 1 (perfect positive correlation).\n- Measures only linear relationships.\n- Useful to quantify strength and direction of linear association.\n### After 1 hour corelation, 12 th row after","metadata":{}},{"cell_type":"markdown","source":"## Spearman Correlation\n\nSpearman's rank correlation measures the **monotonic** relationship between two variables using ranks.\n\n- Value ranges from -1 to 1.\n- Detects monotonic but not necessarily linear relationships.\n- Less sensitive to outliers and works with ordinal data.\n","metadata":{}},{"cell_type":"code","source":"pd.plotting.lag_plot(result[to_show]['label'],lag=13)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:21.891233Z","iopub.execute_input":"2025-07-27T02:26:21.891557Z","iopub.status.idle":"2025-07-27T02:26:22.988574Z","shell.execute_reply.started":"2025-07-27T02:26:21.891528Z","shell.execute_reply":"2025-07-27T02:26:22.987617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result[to_show].index\ndf_reset = result[to_show].reset_index(drop=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:22.989257Z","iopub.execute_input":"2025-07-27T02:26:22.989497Z","iopub.status.idle":"2025-07-27T02:26:24.173008Z","shell.execute_reply.started":"2025-07-27T02:26:22.989480Z","shell.execute_reply":"2025-07-27T02:26:24.172159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_reset.tail(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:24.173860Z","iopub.execute_input":"2025-07-27T02:26:24.174189Z","iopub.status.idle":"2025-07-27T02:26:24.193828Z","shell.execute_reply.started":"2025-07-27T02:26:24.174167Z","shell.execute_reply":"2025-07-27T02:26:24.192862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"multi_data = df_reset[features_main]\nmulti_data.plot(subplots=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:24.194842Z","iopub.execute_input":"2025-07-27T02:26:24.195189Z","iopub.status.idle":"2025-07-27T02:26:25.625785Z","shell.execute_reply.started":"2025-07-27T02:26:24.195160Z","shell.execute_reply":"2025-07-27T02:26:25.624816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_reset.tail(1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:25.626693Z","iopub.execute_input":"2025-07-27T02:26:25.627029Z","iopub.status.idle":"2025-07-27T02:26:25.647717Z","shell.execute_reply.started":"2025-07-27T02:26:25.626997Z","shell.execute_reply":"2025-07-27T02:26:25.646730Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_reset.index","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:25.648709Z","iopub.execute_input":"2025-07-27T02:26:25.649006Z","iopub.status.idle":"2025-07-27T02:26:25.667194Z","shell.execute_reply.started":"2025-07-27T02:26:25.648986Z","shell.execute_reply":"2025-07-27T02:26:25.666220Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"> Both bid_qty and ask_qty are in direct relationship ","metadata":{}},{"cell_type":"code","source":"df_reset.loc['2023-03':'2024-02', ['bid_qty', 'ask_qty']].plot(figsize=(15,8), linewidth=3, fontsize=15)\n\nimport matplotlib.pyplot as plt\nplt.xlabel('year_month', fontsize=20)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:25.668249Z","iopub.execute_input":"2025-07-27T02:26:25.668545Z","iopub.status.idle":"2025-07-27T02:26:25.964333Z","shell.execute_reply.started":"2025-07-27T02:26:25.668514Z","shell.execute_reply":"2025-07-27T02:26:25.963334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_reset.isnull().any()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:25.965246Z","iopub.execute_input":"2025-07-27T02:26:25.965521Z","iopub.status.idle":"2025-07-27T02:26:26.214680Z","shell.execute_reply.started":"2025-07-27T02:26:25.965491Z","shell.execute_reply":"2025-07-27T02:26:26.213864Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"No null values","metadata":{}},{"cell_type":"code","source":"# Get the count of nulls in each column\nnull_counts = df_reset.isnull().sum()\n\n# Filter columns where count of nulls > 0\ncols_with_nulls = null_counts[null_counts > 0]\n\nprint(cols_with_nulls)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:26.215871Z","iopub.execute_input":"2025-07-27T02:26:26.216238Z","iopub.status.idle":"2025-07-27T02:26:26.522128Z","shell.execute_reply.started":"2025-07-27T02:26:26.216207Z","shell.execute_reply":"2025-07-27T02:26:26.521064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"g = sns.pairplot(df_reset[features_main])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:26.523052Z","iopub.execute_input":"2025-07-27T02:26:26.523390Z","iopub.status.idle":"2025-07-27T02:26:53.935508Z","shell.execute_reply.started":"2025-07-27T02:26:26.523346Z","shell.execute_reply":"2025-07-27T02:26:53.934390Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### To check the strength of the relationship between `buy_qty`, `sell_qty`, and `volume`, we use the `.corr()` method. This calculates the **Pearson correlation coefficient** by default, which measures the linear association between numeric columns.","metadata":{}},{"cell_type":"code","source":"df_resets=df_reset[features_main].corr(method='pearson')\ndf_resets","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:53.936586Z","iopub.execute_input":"2025-07-27T02:26:53.936927Z","iopub.status.idle":"2025-07-27T02:26:53.963866Z","shell.execute_reply.started":"2025-07-27T02:26:53.936899Z","shell.execute_reply":"2025-07-27T02:26:53.963098Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### **Interpretation:**\n- Values **close to 1** indicate a strong positive correlation: when one variable is high, the other tends to be high.\n- Values **close to 0** indicate little or no linear relationship.\n- **High correlation between volume and buy/sell quantities** is expected since volume is often derived from the sum of buy and sell trades.\n- When high values are observed, it suggests these fields move together, possibly pointing to market activity spikes.\n\n#### **Visualizing High Values**","metadata":{}},{"cell_type":"code","source":"g = sns.heatmap(df_resets,  vmax=.6, center=0,\n            square=True, linewidths=.5, cbar_kws={\"shrink\": .5}, annot=True, fmt='.2f', cmap='coolwarm')\ng.figure.set_size_inches(10,10)\n    \nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:26:53.964716Z","iopub.execute_input":"2025-07-27T02:26:53.965197Z","iopub.status.idle":"2025-07-27T02:26:54.239711Z","shell.execute_reply.started":"2025-07-27T02:26:53.965163Z","shell.execute_reply":"2025-07-27T02:26:54.238773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_resets.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:30:36.519774Z","iopub.execute_input":"2025-07-27T02:30:36.520160Z","iopub.status.idle":"2025-07-27T02:30:36.526488Z","shell.execute_reply.started":"2025-07-27T02:30:36.520133Z","shell.execute_reply":"2025-07-27T02:30:36.525593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter for March 2023\nmarch1 = result[to_show].loc['2023-03-01':'2023-03-02']['label']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:36:46.615605Z","iopub.execute_input":"2025-07-27T02:36:46.615953Z","iopub.status.idle":"2025-07-27T02:36:48.117241Z","shell.execute_reply.started":"2025-07-27T02:36:46.615928Z","shell.execute_reply":"2025-07-27T02:36:48.116435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.plotting.autocorrelation_plot(march1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-27T02:37:06.654093Z","iopub.execute_input":"2025-07-27T02:37:06.654469Z","iopub.status.idle":"2025-07-27T02:37:06.854300Z","shell.execute_reply.started":"2025-07-27T02:37:06.654445Z","shell.execute_reply":"2025-07-27T02:37:06.853255Z"}},"outputs":[],"execution_count":null}]}