{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\ndf=pd.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the column names with null values\nnull_columns = df.columns[df.isnull().any()]\nprint(null_columns.tolist())\nprint(df.columns)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display the null counts in chunks of 20 columns\nnull_counts = df.isnull().sum()\nfor i in range(0, len(null_counts), 20):\n    print(null_counts[i:i+20])\n    print(\"\\n\" + \"-\"*50 + \"\\n\")  # Add a separator for readability\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Calculate the null count for each column\nnull_counts = df.isnull().sum()\n\n# Plot the null value distribution as a bar chart\nplt.figure(figsize=(12, 8))\nnull_counts[null_counts > 0].sort_values(ascending=False).plot(kind='bar', color='skyblue')\nplt.title(\"Distribution of Null Values Across Columns\")\nplt.xlabel(\"Columns\")\nplt.ylabel(\"Number of Null Values\")\nplt.xticks(rotation=90)\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Identify columns with missing values\nnull_columns = df.columns[df.isnull().any()]\nprint(\"Columns with null values:\", null_columns.tolist())\n\n# Select the first two columns with null values\ncolumns_to_plot = null_columns\n\n# Plot distributions for the selected columns\nfor col in columns_to_plot:\n    print(col)\n    plt.figure(figsize=(8, 4))\n    df[col].dropna().hist(bins=10, color='skyblue', edgecolor='black')  # Drop nulls for plotting\n    plt.title(f\"Distribution of {col}\")\n    plt.xlabel(col)\n    plt.ylabel(\"Frequency\")\n    plt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df.head(20))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport torch\nimport numpy as np\n\ndf = pd.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4\")\n\nnull_columns = df.columns[df.isnull().any()]\nprint(\"Columns with NULL values:\", null_columns.tolist())\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(device)\ndef rolling_median_fill(tensor_data, window_size=1000):\n    filled_data = tensor_data.clone()\n    \n    for i in range(len(tensor_data)):\n        if torch.isnan(tensor_data[i]):\n            start = max(0, i - window_size + 1)\n            end = i + 1\n            \n            window_values = tensor_data[start:end]\n            window_values = window_values[~torch.isnan(window_values)]\n            \n            if len(window_values) > 0:\n                filled_data[i] = torch.median(window_values)\n    \n    return filled_data\n\nfor col in null_columns:\n    data_tensor = torch.tensor(df[col].values, dtype=torch.float32, device=device)\n    filled_tensor = rolling_median_fill(data_tensor, window_size=10000)\n    \n    col_median = torch.median(filled_tensor[~torch.isnan(filled_tensor)])\n    filled_tensor = torch.where(torch.isnan(filled_tensor), col_median, filled_tensor)\n    \n    df[col] = filled_tensor.cpu().numpy()\n\nprint(\"Null values filled.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-06T17:31:21.438416Z","iopub.execute_input":"2024-11-06T17:31:21.438809Z","iopub.status.idle":"2024-11-06T19:03:42.505751Z","shell.execute_reply.started":"2024-11-06T17:31:21.438768Z","shell.execute_reply":"2024-11-06T19:03:42.504615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the modified DataFrame to a new Parquet file\ndf.to_parquet(\"/kaggle/working/filled_train.parquet\")\nprint(\"DataFrame saved as filled_train.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-06T19:03:42.507814Z","iopub.execute_input":"2024-11-06T19:03:42.509033Z","iopub.status.idle":"2024-11-06T19:04:03.181745Z","shell.execute_reply.started":"2024-11-06T19:03:42.508975Z","shell.execute_reply":"2024-11-06T19:04:03.180746Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}