{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:41:45.579118Z","iopub.execute_input":"2025-06-25T08:41:45.579929Z","iopub.status.idle":"2025-06-25T08:41:45.587851Z","shell.execute_reply.started":"2025-06-25T08:41:45.579894Z","shell.execute_reply":"2025-06-25T08:41:45.586654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# DRW Crypto Market Prediction - Starter Notebook\n\nimport pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.feature_selection import SelectKBest, f_regression\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.impute import SimpleImputer\nimport os\nfrom lightgbm import LGBMRegressor, early_stopping\n\n# --- Load Data (Parquet) ---\ntrain_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\ntest_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\nsample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n# Ensure file paths exist\nassert os.path.exists(train_path), \"Train parquet file not found.\"\nassert os.path.exists(test_path), \"Test parquet file not found.\"\nassert os.path.exists(sample_sub_path), \"Sample submission file not found.\"\n\ntrain = pd.read_parquet(train_path)\ntest = pd.read_parquet(test_path)\nsample_submission = pd.read_csv(sample_sub_path)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:00:01.999863Z","iopub.execute_input":"2025-06-25T08:00:02.000476Z","iopub.status.idle":"2025-06-25T08:00:57.746515Z","shell.execute_reply.started":"2025-06-25T08:00:02.000443Z","shell.execute_reply":"2025-06-25T08:00:57.745229Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"column_names = train.columns.tolist()\nprint(column_names)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:01:15.145842Z","iopub.execute_input":"2025-06-25T08:01:15.146366Z","iopub.status.idle":"2025-06-25T08:01:15.152883Z","shell.execute_reply.started":"2025-06-25T08:01:15.146328Z","shell.execute_reply":"2025-06-25T08:01:15.151957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:01:18.909886Z","iopub.execute_input":"2025-06-25T08:01:18.910286Z","iopub.status.idle":"2025-06-25T08:01:18.950803Z","shell.execute_reply.started":"2025-06-25T08:01:18.910256Z","shell.execute_reply":"2025-06-25T08:01:18.949634Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:01:21.867959Z","iopub.execute_input":"2025-06-25T08:01:21.868366Z","iopub.status.idle":"2025-06-25T08:01:21.896382Z","shell.execute_reply.started":"2025-06-25T08:01:21.868337Z","shell.execute_reply":"2025-06-25T08:01:21.895209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Features and Target ---\ntarget = train[\"label\"]\nfeatures = train.drop(columns=[\"label\"])\ntest_features = test#.drop(columns=[\"timestamp_id\"])\n# --- Features and Target ---\n\n# --- Clean Data: Replace inf/-inf with NaN, then fill or drop ---\nfeatures.replace([np.inf, -np.inf], np.nan, inplace=True)\ntest_features.replace([np.inf, -np.inf], np.nan, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:01:24.536405Z","iopub.execute_input":"2025-06-25T08:01:24.536742Z","iopub.status.idle":"2025-06-25T08:01:34.999781Z","shell.execute_reply.started":"2025-06-25T08:01:24.536715Z","shell.execute_reply":"2025-06-25T08:01:34.998575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"common_columns = features.columns.intersection(test_features.columns)\nfeatures = features[common_columns]\ntest_features = test_features[common_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:04:24.360100Z","iopub.execute_input":"2025-06-25T08:04:24.360632Z","iopub.status.idle":"2025-06-25T08:04:31.163366Z","shell.execute_reply.started":"2025-06-25T08:04:24.360586Z","shell.execute_reply":"2025-06-25T08:04:31.162251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#imputer = SimpleImputer(strategy=\"median\")\nimputer = SimpleImputer(strategy=\"constant\", fill_value=0)\nfeatures_imputed = imputer.fit_transform(features)\ntest_imputed = imputer.transform(test_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:04:33.787791Z","iopub.execute_input":"2025-06-25T08:04:33.788739Z","iopub.status.idle":"2025-06-25T08:05:08.866810Z","shell.execute_reply.started":"2025-06-25T08:04:33.788703Z","shell.execute_reply":"2025-06-25T08:05:08.864443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Feature Selection (optional) ---\nselector = SelectKBest(score_func=f_regression, k=50)\nfeatures_selected = selector.fit_transform(features_imputed, target)\ntest_selected = selector.transform(test_imputed)\n\n# --- Model Training ---\ntscv = TimeSeriesSplit(n_splits=5)\npreds = np.zeros(len(test))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:05:18.872925Z","iopub.execute_input":"2025-06-25T08:05:18.873371Z","iopub.status.idle":"2025-06-25T08:05:22.984457Z","shell.execute_reply.started":"2025-06-25T08:05:18.873337Z","shell.execute_reply":"2025-06-25T08:05:22.983285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\npreds = np.zeros(len(test_selected))\n\nfor fold, (train_idx, val_idx) in enumerate(tscv.split(features_selected)):\n    X_train, X_val = features_selected[train_idx], features_selected[val_idx]\n    y_train, y_val = target.iloc[train_idx], target.iloc[val_idx]\n\n    model = LGBMRegressor(\n        n_estimators=1000,\n        learning_rate=0.01,\n        num_leaves=64,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        random_state=42\n    )\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_val, y_val)],\n        callbacks=[early_stopping(stopping_rounds=50, verbose=True)]\n    )\n\n    preds += model.predict(test_selected) / tscv.get_n_splits()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:09:12.232866Z","iopub.execute_input":"2025-06-25T08:09:12.234018Z","iopub.status.idle":"2025-06-25T08:09:40.327908Z","shell.execute_reply.started":"2025-06-25T08:09:12.233952Z","shell.execute_reply":"2025-06-25T08:09:40.326875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:41:36.398700Z","iopub.execute_input":"2025-06-25T08:41:36.399598Z","iopub.status.idle":"2025-06-25T08:41:36.410686Z","shell.execute_reply.started":"2025-06-25T08:41:36.399566Z","shell.execute_reply":"2025-06-25T08:41:36.409677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Create Submission ---\nsample_submission[\"prediction\"] = preds\nsample_submission.to_csv(\"submission.csv\", index=False)\n\n\nprint(\"Submission file saved as submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T08:41:23.364196Z","iopub.execute_input":"2025-06-25T08:41:23.365224Z","iopub.status.idle":"2025-06-25T08:41:25.758418Z","shell.execute_reply.started":"2025-06-25T08:41:23.365189Z","shell.execute_reply":"2025-06-25T08:41:25.757167Z"}},"outputs":[],"execution_count":null}]}