{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:34:41.423704Z","iopub.execute_input":"2025-06-25T11:34:41.424788Z","iopub.status.idle":"2025-06-25T11:34:42.704025Z","shell.execute_reply.started":"2025-06-25T11:34:41.42474Z","shell.execute_reply":"2025-06-25T11:34:42.702955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport gc\nfrom sklearn.cluster import KMeans\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler\nfrom lightgbm import LGBMRegressor, early_stopping, log_evaluation","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:42:35.869747Z","iopub.execute_input":"2025-06-25T11:42:35.870264Z","iopub.status.idle":"2025-06-25T11:42:35.875869Z","shell.execute_reply.started":"2025-06-25T11:42:35.870145Z","shell.execute_reply":"2025-06-25T11:42:35.874926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Load Data (Parquet) ---\ntrain_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\ntest_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\nsample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n\n# Ensure file paths exist\nassert os.path.exists(train_path), \"Train parquet file not found.\"\nassert os.path.exists(test_path), \"Test parquet file not found.\"\nassert os.path.exists(sample_sub_path), \"Sample submission file not found.\"\n\ndf_full = pd.read_parquet(train_path)\ntest = pd.read_parquet(test_path)\nsample_submission = pd.read_csv(sample_sub_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:35:34.623186Z","iopub.execute_input":"2025-06-25T11:35:34.623954Z","iopub.status.idle":"2025-06-25T11:36:26.735227Z","shell.execute_reply.started":"2025-06-25T11:35:34.623919Z","shell.execute_reply":"2025-06-25T11:36:26.733755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_full.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:36:46.464673Z","iopub.execute_input":"2025-06-25T11:36:46.465041Z","iopub.status.idle":"2025-06-25T11:36:46.50368Z","shell.execute_reply.started":"2025-06-25T11:36:46.465014Z","shell.execute_reply":"2025-06-25T11:36:46.502785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"type(df_full)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:25:53.877665Z","iopub.execute_input":"2025-06-25T11:25:53.878669Z","iopub.status.idle":"2025-06-25T11:25:53.884595Z","shell.execute_reply.started":"2025-06-25T11:25:53.878634Z","shell.execute_reply":"2025-06-25T11:25:53.883788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_full.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:18:30.250271Z","iopub.execute_input":"2025-06-25T11:18:30.250628Z","iopub.status.idle":"2025-06-25T11:18:30.258006Z","shell.execute_reply.started":"2025-06-25T11:18:30.250598Z","shell.execute_reply":"2025-06-25T11:18:30.256802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_full = df_full.reset_index()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:36:52.022712Z","iopub.execute_input":"2025-06-25T11:36:52.023424Z","iopub.status.idle":"2025-06-25T11:36:53.689657Z","shell.execute_reply.started":"2025-06-25T11:36:52.023365Z","shell.execute_reply":"2025-06-25T11:36:53.688107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === STEP 1: Filter to last year ===\ndf_full[\"timestamp\"] = pd.to_datetime(df_full[\"timestamp\"])  # ensure datetime\nlatest_year = df_full[\"timestamp\"].dt.year.max()\ndf_last_year = df_full[df_full[\"timestamp\"].dt.year == latest_year].reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:37:14.122458Z","iopub.execute_input":"2025-06-25T11:37:14.122862Z","iopub.status.idle":"2025-06-25T11:37:14.836547Z","shell.execute_reply.started":"2025-06-25T11:37:14.12283Z","shell.execute_reply":"2025-06-25T11:37:14.835492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reduce_mem_usage(dataframe):    \n    print('Reducing memory usage for:', dataframe)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n\n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:37:17.33629Z","iopub.execute_input":"2025-06-25T11:37:17.33667Z","iopub.status.idle":"2025-06-25T11:37:17.348049Z","shell.execute_reply.started":"2025-06-25T11:37:17.336638Z","shell.execute_reply":"2025-06-25T11:37:17.34652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test =  reduce_mem_usage(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:37:22.122507Z","iopub.execute_input":"2025-06-25T11:37:22.122914Z","iopub.status.idle":"2025-06-25T11:37:30.449997Z","shell.execute_reply.started":"2025-06-25T11:37:22.122882Z","shell.execute_reply":"2025-06-25T11:37:30.449057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Separate features and target\ntarget_col = \"label\"  # change to your actual target column name\ndrop_cols = ['timestamp', target_col]  # columns to exclude from features\n\nfeatures = df_last_year.drop(columns=drop_cols)\ntarget = df_last_year[target_col]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:37:35.422439Z","iopub.execute_input":"2025-06-25T11:37:35.422807Z","iopub.status.idle":"2025-06-25T11:37:35.625674Z","shell.execute_reply.started":"2025-06-25T11:37:35.422778Z","shell.execute_reply":"2025-06-25T11:37:35.624789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_last_year.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:37:38.422003Z","iopub.execute_input":"2025-06-25T11:37:38.422346Z","iopub.status.idle":"2025-06-25T11:37:38.448907Z","shell.execute_reply.started":"2025-06-25T11:37:38.42232Z","shell.execute_reply":"2025-06-25T11:37:38.447769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = reduce_mem_usage(features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:37:41.491127Z","iopub.execute_input":"2025-06-25T11:37:41.491528Z","iopub.status.idle":"2025-06-25T11:37:43.015154Z","shell.execute_reply.started":"2025-06-25T11:37:41.4915Z","shell.execute_reply":"2025-06-25T11:37:43.014271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = features.replace([np.inf, -np.inf], np.nan)\ntest = test.replace([np.inf, -np.inf], np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:37:46.415405Z","iopub.execute_input":"2025-06-25T11:37:46.415765Z","iopub.status.idle":"2025-06-25T11:37:53.820867Z","shell.execute_reply.started":"2025-06-25T11:37:46.415737Z","shell.execute_reply":"2025-06-25T11:37:53.81985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"common_columns = features.columns.intersection(test.columns)\nfeatures = features[common_columns]\ntest = test[common_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:38:12.428593Z","iopub.execute_input":"2025-06-25T11:38:12.429094Z","iopub.status.idle":"2025-06-25T11:38:14.447182Z","shell.execute_reply.started":"2025-06-25T11:38:12.429049Z","shell.execute_reply":"2025-06-25T11:38:14.446074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === STEP 2: Impute missing values ===\nfrom sklearn.impute import SimpleImputer\n\nimputer = SimpleImputer(strategy=\"constant\", fill_value=0)\nfeatures_imputed = imputer.fit_transform(features)\n\n\ntest_imputed = imputer.transform(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:38:17.017124Z","iopub.execute_input":"2025-06-25T11:38:17.017444Z","iopub.status.idle":"2025-06-25T11:38:34.446162Z","shell.execute_reply.started":"2025-06-25T11:38:17.017422Z","shell.execute_reply":"2025-06-25T11:38:34.445174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\n# Clean first\nfeatures_imputed = np.nan_to_num(features_imputed, nan=0.0, posinf=0.0, neginf=0.0)\ntest_imputed = np.nan_to_num(test_imputed, nan=0.0, posinf=0.0, neginf=0.0)\n\n# Optional: clip huge outliers\nfeatures_imputed = np.clip(features_imputed, -1e6, 1e6)\ntest_imputed = np.clip(test_imputed, -1e6, 1e6)\n\n# Now scale safely\nscaler = StandardScaler()\nfeatures_scaled = scaler.fit_transform(features_imputed)\ntest_scaled = scaler.transform(test_imputed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:38:41.592345Z","iopub.execute_input":"2025-06-25T11:38:41.59269Z","iopub.status.idle":"2025-06-25T11:39:01.733986Z","shell.execute_reply.started":"2025-06-25T11:38:41.592663Z","shell.execute_reply":"2025-06-25T11:39:01.732825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# === STEP 4: Clustering ===\nn_clusters = 5\nkmeans = KMeans(n_clusters=n_clusters, random_state=42)\ncluster_labels_train = kmeans.fit_predict(features_scaled)\ncluster_labels_test = kmeans.predict(test_scaled)\n\n# Add cluster labels\nfeatures_df = pd.DataFrame(features_imputed, columns=features.columns)\ntest_df = pd.DataFrame(test_imputed, columns=features.columns)\n\nfeatures_df[\"cluster\"] = cluster_labels_train\ntest_df[\"cluster\"] = cluster_labels_test\n\nX = features_df.values\nX_test = test_df.values\ny = target.values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:39:08.459363Z","iopub.execute_input":"2025-06-25T11:39:08.45975Z","iopub.status.idle":"2025-06-25T11:40:27.457459Z","shell.execute_reply.started":"2025-06-25T11:39:08.459706Z","shell.execute_reply":"2025-06-25T11:40:27.456381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del features_imputed, test_imputed, scaler\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:42:44.387541Z","iopub.execute_input":"2025-06-25T11:42:44.387918Z","iopub.status.idle":"2025-06-25T11:42:44.408472Z","shell.execute_reply.started":"2025-06-25T11:42:44.38789Z","shell.execute_reply":"2025-06-25T11:42:44.407074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === STEP 5: TimeSeries CV with memory-safe training ===\npreds = np.zeros(X_test.shape[0])\ntscv = TimeSeriesSplit(n_splits=3)\n\nfor fold, (train_idx, val_idx) in enumerate(tscv.split(X)):\n    print(f\"Fold {fold + 1}\")\n\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n\n    model = LGBMRegressor(\n        n_estimators=500,\n        learning_rate=0.01,\n        num_leaves=64,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        random_state=42\n    )\n\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_val, y_val)],\n        callbacks=[early_stopping(stopping_rounds=30),\n        log_evaluation(0)]\n    )\n\n    preds += model.predict(X_test) / tscv.get_n_splits()\n\n    del model, X_train, X_val, y_train, y_val\n    gc.collect()\n\nprint(\"Training complete. Final predictions ready.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T11:42:46.716499Z","iopub.execute_input":"2025-06-25T11:42:46.71689Z","execution_failed":"2025-06-25T11:43:03.233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# --- Create Submission ---\nsample_submission[\"prediction\"] = preds\nsample_submission.to_csv(\"submission.csv\", index=False)\n\n\nprint(\"Submission file saved as submission.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}