{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:56:04.359349Z","iopub.execute_input":"2025-07-06T10:56:04.359581Z","iopub.status.idle":"2025-07-06T10:56:06.449827Z","shell.execute_reply.started":"2025-07-06T10:56:04.359561Z","shell.execute_reply":"2025-07-06T10:56:06.448982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:56:06.451407Z","iopub.execute_input":"2025-07-06T10:56:06.451775Z","iopub.status.idle":"2025-07-06T10:56:32.149176Z","shell.execute_reply.started":"2025-07-06T10:56:06.451753Z","shell.execute_reply":"2025-07-06T10:56:32.148404Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:56:32.150068Z","iopub.execute_input":"2025-07-06T10:56:32.150396Z","iopub.status.idle":"2025-07-06T10:56:32.191539Z","shell.execute_reply.started":"2025-07-06T10:56:32.150366Z","shell.execute_reply":"2025-07-06T10:56:32.190619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-03T13:32:08.55624Z","iopub.execute_input":"2025-07-03T13:32:08.556604Z","iopub.status.idle":"2025-07-03T13:32:08.565044Z","shell.execute_reply.started":"2025-07-03T13:32:08.556577Z","shell.execute_reply":"2025-07-03T13:32:08.563849Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.plot(df[\"bid_qty\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-04T06:17:34.186223Z","iopub.execute_input":"2025-07-04T06:17:34.186484Z","iopub.status.idle":"2025-07-04T06:17:34.72286Z","shell.execute_reply.started":"2025-07-04T06:17:34.186461Z","shell.execute_reply":"2025-07-04T06:17:34.72197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum().sort_values(ascending = False) # No null values ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-04T06:23:13.316756Z","iopub.execute_input":"2025-07-04T06:23:13.317197Z","iopub.status.idle":"2025-07-04T06:23:15.008198Z","shell.execute_reply.started":"2025-07-04T06:23:13.317163Z","shell.execute_reply":"2025-07-04T06:23:15.007296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df.drop(\"label\", axis = 1)\nY = df[\"label\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:14:27.649223Z","iopub.execute_input":"2025-07-06T10:14:27.650409Z","iopub.status.idle":"2025-07-06T10:14:29.648921Z","shell.execute_reply.started":"2025-07-06T10:14:27.650376Z","shell.execute_reply":"2025-07-06T10:14:29.647487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X is your time-series DataFrame (assumes index is datetime or ordered)\n\nX_imputed = X.copy()\n\n# Replace infs with NaN first if needed\nX_imputed.replace([np.inf, -np.inf], np.nan, inplace=True)\n\n# Apply rolling-window mean imputation with window=11, centered\nX_imputed = X_imputed.apply(lambda col: col.fillna(col.rolling(window=11, center=True, min_periods=1).mean()))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:14:30.6806Z","iopub.execute_input":"2025-07-06T10:14:30.681244Z","iopub.status.idle":"2025-07-06T10:15:00.646365Z","shell.execute_reply.started":"2025-07-06T10:14:30.681216Z","shell.execute_reply":"2025-07-06T10:15:00.645337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nnp.isinf(X_imputed).any()\nnp.isnan(X_imputed).any()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:15:00.648107Z","iopub.execute_input":"2025-07-06T10:15:00.648424Z","iopub.status.idle":"2025-07-06T10:15:03.734487Z","shell.execute_reply.started":"2025-07-06T10:15:00.648399Z","shell.execute_reply":"2025-07-06T10:15:03.733323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compute variance manually\nvariances = X_imputed.var(axis=0)\n\n# Check for any NaNs, infs or very small values in variances\nprint(\"NaN in variance:\", variances.isna().any())\nprint(\"Inf in variance:\", np.isinf(variances).any())\nprint(\"Zero variance columns:\", (variances == 0).sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:15:03.735518Z","iopub.execute_input":"2025-07-06T10:15:03.735842Z","iopub.status.idle":"2025-07-06T10:15:05.495543Z","shell.execute_reply.started":"2025-07-06T10:15:03.73582Z","shell.execute_reply":"2025-07-06T10:15:05.494088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Drop NaN-variance columns\nvariances = X_imputed.var(axis=0)\nvalid_variance_mask = ~variances.isna()\nX_valid = X_imputed.loc[:, valid_variance_mask]\n\n# Step 2: Recompute variance only on clean data\nclean_variances = X_valid.var(axis=0)\nprint(\"NaN in variance:\", clean_variances.isna().any())\nprint(\"Inf in variance:\", np.isinf(clean_variances).any())\nprint(\"Zero variance columns:\", (clean_variances == 0).sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:15:05.497904Z","iopub.execute_input":"2025-07-06T10:15:05.498196Z","iopub.status.idle":"2025-07-06T10:15:10.79328Z","shell.execute_reply.started":"2025-07-06T10:15:05.498174Z","shell.execute_reply":"2025-07-06T10:15:10.792172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.feature_selection import VarianceThreshold\nselector = VarianceThreshold(threshold=0.01)  # or 0.0 to drop constant\nX_reduced = selector.fit_transform(X_imputed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:15:10.794411Z","iopub.execute_input":"2025-07-06T10:15:10.794865Z","iopub.status.idle":"2025-07-06T10:15:26.096296Z","shell.execute_reply.started":"2025-07-06T10:15:10.794827Z","shell.execute_reply":"2025-07-06T10:15:26.095073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Removing ","metadata":{}},{"cell_type":"code","source":"X_reduced","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:15:26.098283Z","iopub.execute_input":"2025-07-06T10:15:26.098879Z","iopub.status.idle":"2025-07-06T10:15:26.107053Z","shell.execute_reply.started":"2025-07-06T10:15:26.098829Z","shell.execute_reply":"2025-07-06T10:15:26.105975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_reduced.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:22:44.607138Z","iopub.execute_input":"2025-07-06T10:22:44.607521Z","iopub.status.idle":"2025-07-06T10:22:44.61457Z","shell.execute_reply.started":"2025-07-06T10:22:44.607497Z","shell.execute_reply":"2025-07-06T10:22:44.613317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:17:55.068243Z","iopub.execute_input":"2025-07-06T10:17:55.068609Z","iopub.status.idle":"2025-07-06T10:17:55.076883Z","shell.execute_reply.started":"2025-07-06T10:17:55.068587Z","shell.execute_reply":"2025-07-06T10:17:55.07571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\ncorr_matrix = pd.DataFrame(X_reduced).corr().abs()\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n\n# Drop one of each pair of correlated features\nto_drop = [column for column in upper.columns if any(upper[column] > 0.9)]\nX_filtered = pd.DataFrame(X).drop(columns=to_drop)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.loc[:, ~df.isin([np.inf, -np.inf]).any()]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:56:32.192339Z","iopub.execute_input":"2025-07-06T10:56:32.192563Z","iopub.status.idle":"2025-07-06T10:57:32.463646Z","shell.execute_reply.started":"2025-07-06T10:56:32.192544Z","shell.execute_reply":"2025-07-06T10:57:32.462752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_new = df.drop(\"label\", axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:57:32.464743Z","iopub.execute_input":"2025-07-06T10:57:32.465051Z","iopub.status.idle":"2025-07-06T10:57:33.931215Z","shell.execute_reply.started":"2025-07-06T10:57:32.465015Z","shell.execute_reply":"2025-07-06T10:57:33.930286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_new.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T10:57:33.941958Z","iopub.execute_input":"2025-07-06T10:57:33.942628Z","iopub.status.idle":"2025-07-06T10:57:33.957973Z","shell.execute_reply.started":"2025-07-06T10:57:33.942597Z","shell.execute_reply":"2025-07-06T10:57:33.957344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\ncorr_matrix = pd.DataFrame(X_new).corr().abs()\nupper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n\n# Drop one of each pair of correlated features\nto_drop = [column for column in upper.columns if any(upper[column] > 0.9)]\nX_filtered = pd.DataFrame(X_new).drop(columns=to_drop)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T11:16:49.591435Z","iopub.execute_input":"2025-07-06T11:16:49.591698Z","iopub.status.idle":"2025-07-06T11:34:47.755926Z","shell.execute_reply.started":"2025-07-06T11:16:49.591679Z","shell.execute_reply":"2025-07-06T11:34:47.755144Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_filtered.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T11:36:27.805226Z","iopub.execute_input":"2025-07-06T11:36:27.805572Z","iopub.status.idle":"2025-07-06T11:36:27.812922Z","shell.execute_reply.started":"2025-07-06T11:36:27.805541Z","shell.execute_reply":"2025-07-06T11:36:27.812067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\n\n# Step 1: Scale the data (zero mean, unit variance)\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X_filtered)  # <-- X_filtered = cleaned data\n\n# Step 2: Fit PCA\npca = PCA(n_components=0.95)  # Retain 95% of variance\nX_pca = pca.fit_transform(X_scaled)\n\n# Optional: View explained variance ratio\nimport matplotlib.pyplot as plt\nplt.plot(np.cumsum(pca.explained_variance_ratio_))\nplt.xlabel(\"Number of Components\")\nplt.ylabel(\"Cumulative Explained Variance\")\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T11:37:39.480272Z","iopub.execute_input":"2025-07-06T11:37:39.481061Z","iopub.status.idle":"2025-07-06T11:38:12.05935Z","shell.execute_reply.started":"2025-07-06T11:37:39.481038Z","shell.execute_reply":"2025-07-06T11:38:12.058392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_pca.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T11:38:47.340615Z","iopub.execute_input":"2025-07-06T11:38:47.341515Z","iopub.status.idle":"2025-07-06T11:38:47.346893Z","shell.execute_reply.started":"2025-07-06T11:38:47.341484Z","shell.execute_reply":"2025-07-06T11:38:47.346198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor, VotingRegressor\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.preprocessing import StandardScaler\nimport numpy as np\n\n# Step 1: Use the processed PCA DataFrame\ndf_model = df_pca_combined.copy()  # this is your PCA+known+label DataFrame\n\n# Step 2: Separate features and label\nX = df_model.drop(columns=['label'])\ny = df_model['label']\n\n# (Optional: all features are already clean and standardized, so no need to rescale)\n\n# Step 3: Train-test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Step 4: Define base regressors\nrf = RandomForestRegressor(n_estimators=100, random_state=42)\ngb = GradientBoostingRegressor(n_estimators=100, learning_rate=0.1, random_state=42)\nlr = LinearRegression()\n\n# Step 5: Ensemble Regressor\nensemble = VotingRegressor(estimators=[\n    ('rf', rf),\n    ('gb', gb),\n    ('lr', lr)\n])\n\n# Step 6: Train and evaluate\nensemble.fit(X_train, y_train)\ny_pred = ensemble.predict(X_test)\n\nprint(\"RMSE:\", np.sqrt(mean_squared_error(y_test, y_pred)))\nprint(\"R² score:\", r2_score(y_test, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-04T06:46:22.933686Z","iopub.status.idle":"2025-07-04T06:46:22.934106Z","shell.execute_reply.started":"2025-07-04T06:46:22.933935Z","shell.execute_reply":"2025-07-04T06:46:22.933952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}