{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:00:49.120039Z","iopub.execute_input":"2025-06-25T10:00:49.122438Z","iopub.status.idle":"2025-06-25T10:00:50.66968Z","shell.execute_reply.started":"2025-06-25T10:00:49.122341Z","shell.execute_reply":"2025-06-25T10:00:50.668583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.feature_selection import SelectKBest, f_regression\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.impute import SimpleImputer\nimport os\nfrom lightgbm import LGBMRegressor, early_stopping\n\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import StandardScaler\n\n\n# --- Load Data (Parquet) ---\ntrain_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\ntest_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\nsample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n# Ensure file paths exist\nassert os.path.exists(train_path), \"Train parquet file not found.\"\nassert os.path.exists(test_path), \"Test parquet file not found.\"\nassert os.path.exists(sample_sub_path), \"Sample submission file not found.\"\n\ntrain = pd.read_parquet(train_path)\ntest = pd.read_parquet(test_path)\nsample_submission = pd.read_csv(sample_sub_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:00:54.920915Z","iopub.execute_input":"2025-06-25T10:00:54.921408Z","iopub.status.idle":"2025-06-25T10:02:01.028367Z","shell.execute_reply.started":"2025-06-25T10:00:54.921377Z","shell.execute_reply":"2025-06-25T10:02:01.026839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(dataframe):    \n    print('Reducing memory usage for:', dataframe)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n\n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:06:13.748277Z","iopub.execute_input":"2025-06-25T10:06:13.748675Z","iopub.status.idle":"2025-06-25T10:06:13.761085Z","shell.execute_reply.started":"2025-06-25T10:06:13.748646Z","shell.execute_reply":"2025-06-25T10:06:13.759985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = reduce_mem_usage(train)\ntest =  reduce_mem_usage(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:06:18.321003Z","iopub.execute_input":"2025-06-25T10:06:18.321354Z","iopub.status.idle":"2025-06-25T10:06:35.726421Z","shell.execute_reply.started":"2025-06-25T10:06:18.32133Z","shell.execute_reply":"2025-06-25T10:06:35.725147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Features and Target ---\ntarget = train[\"label\"]\nfeatures = train.drop(columns=[\"label\"])\ntest_features = test#.drop(columns=[\"timestamp_id\"])\n# --- Features and Target ---\n\n# --- Clean Data: Replace inf/-inf with NaN, then fill or drop ---\nfeatures.replace([np.inf, -np.inf], np.nan, inplace=True)\ntest_features.replace([np.inf, -np.inf], np.nan, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:06:40.885723Z","iopub.execute_input":"2025-06-25T10:06:40.886098Z","iopub.status.idle":"2025-06-25T10:06:53.324256Z","shell.execute_reply.started":"2025-06-25T10:06:40.886069Z","shell.execute_reply":"2025-06-25T10:06:53.323219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"common_columns = features.columns.intersection(test_features.columns)\nfeatures = features[common_columns]\ntest_features = test_features[common_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:06:57.384993Z","iopub.execute_input":"2025-06-25T10:06:57.385344Z","iopub.status.idle":"2025-06-25T10:07:01.009299Z","shell.execute_reply.started":"2025-06-25T10:06:57.385319Z","shell.execute_reply":"2025-06-25T10:07:01.008099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer = SimpleImputer(strategy=\"constant\", fill_value=0)\nfeatures_imputed = imputer.fit_transform(features)\ntest_imputed = imputer.transform(test_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:07:09.668988Z","iopub.execute_input":"2025-06-25T10:07:09.669314Z","iopub.status.idle":"2025-06-25T10:07:44.553927Z","shell.execute_reply.started":"2025-06-25T10:07:09.669292Z","shell.execute_reply":"2025-06-25T10:07:44.55291Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features_scaled = features_imputed\ntest_scaled = test_imputed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:08:20.133263Z","iopub.execute_input":"2025-06-25T10:08:20.133635Z","iopub.status.idle":"2025-06-25T10:08:20.138306Z","shell.execute_reply.started":"2025-06-25T10:08:20.133608Z","shell.execute_reply":"2025-06-25T10:08:20.137348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_clusters = 5\nkmeans = KMeans(n_clusters=n_clusters, random_state=42)\ncluster_labels_train = kmeans.fit_predict(features_scaled)\ncluster_labels_test = kmeans.predict(test_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:08:51.744731Z","iopub.execute_input":"2025-06-25T10:08:51.745129Z","iopub.status.idle":"2025-06-25T10:14:05.147454Z","shell.execute_reply.started":"2025-06-25T10:08:51.745095Z","shell.execute_reply":"2025-06-25T10:14:05.146439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add cluster label as new feature\nfeatures_clustered = np.hstack([features_scaled, cluster_labels_train.reshape(-1, 1)])\ntest_clustered = np.hstack([test_imputed, cluster_labels_test.reshape(-1, 1)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:19:36.236739Z","iopub.execute_input":"2025-06-25T10:19:36.237229Z","iopub.status.idle":"2025-06-25T10:19:36.251838Z","shell.execute_reply.started":"2025-06-25T10:19:36.237188Z","shell.execute_reply":"2025-06-25T10:19:36.250262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize prediction array\npreds = np.zeros(len(test_imputed))\n\n\ntscv = TimeSeriesSplit(n_splits=5)\n# CV\nfor fold, (train_idx, val_idx) in enumerate(tscv.split(features_clustered)):\n    X_train, X_val = features_clustered[train_idx], features_clustered[val_idx]\n    y_train, y_val = target.iloc[train_idx], target.iloc[val_idx]\n\n    model = LGBMRegressor(\n        n_estimators=1000,\n        learning_rate=0.01,\n        num_leaves=64,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        random_state=42\n    )\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_val, y_val)],\n        callbacks=[early_stopping(stopping_rounds=50, verbose=True)]\n    )\n\n    preds += model.predict(test_clustered) / tscv.get_n_splits()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T10:18:51.883331Z","iopub.status.idle":"2025-06-25T10:18:51.88365Z","shell.execute_reply.started":"2025-06-25T10:18:51.883475Z","shell.execute_reply":"2025-06-25T10:18:51.883504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Create Submission ---\nsample_submission[\"prediction\"] = preds\nsample_submission.to_csv(\"submission.csv\", index=False)\n\n\nprint(\"Submission file saved as submission.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}