{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# DRW Crypto Market Prediction - Advanced Pipeline ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.feature_selection import SelectKBest, f_regression\nfrom sklearn.metrics import mean_squared_error, r2_score\nimport xgboost as xgb\nimport lightgbm as lgb\nimport time\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:00:37.928556Z","iopub.execute_input":"2025-07-06T21:00:37.928849Z","iopub.status.idle":"2025-07-06T21:00:43.909669Z","shell.execute_reply.started":"2025-07-06T21:00:37.928827Z","shell.execute_reply":"2025-07-06T21:00:43.908896Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Data Loading","metadata":{}},{"cell_type":"code","source":"train = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:00:43.910852Z","iopub.execute_input":"2025-07-06T21:00:43.911326Z","iopub.status.idle":"2025-07-06T21:01:24.747614Z","shell.execute_reply.started":"2025-07-06T21:00:43.911307Z","shell.execute_reply":"2025-07-06T21:01:24.746797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape\ntest.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:24.748597Z","iopub.execute_input":"2025-07-06T21:01:24.749045Z","iopub.status.idle":"2025-07-06T21:01:24.754266Z","shell.execute_reply.started":"2025-07-06T21:01:24.749020Z","shell.execute_reply":"2025-07-06T21:01:24.753734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:24.755883Z","iopub.execute_input":"2025-07-06T21:01:24.756140Z","iopub.status.idle":"2025-07-06T21:01:24.798930Z","shell.execute_reply.started":"2025-07-06T21:01:24.756124Z","shell.execute_reply":"2025-07-06T21:01:24.798375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:24.799464Z","iopub.execute_input":"2025-07-06T21:01:24.799718Z","iopub.status.idle":"2025-07-06T21:01:24.943460Z","shell.execute_reply.started":"2025-07-06T21:01:24.799692Z","shell.execute_reply":"2025-07-06T21:01:24.942908Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Optimize Memory","metadata":{}},{"cell_type":"code","source":"def reduce_memory_usage(df):\n    for col in df.columns:\n        col_type = df[col].dtype\n        if col_type == 'float64':\n            df[col] = pd.to_numeric(df[col], downcast='float')\n        elif col_type == 'int64':\n            df[col] = pd.to_numeric(df[col], downcast='integer')\n    return df\n\ntrain = reduce_memory_usage(train)\ntest = reduce_memory_usage(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:24.944141Z","iopub.execute_input":"2025-07-06T21:01:24.944410Z","iopub.status.idle":"2025-07-06T21:01:40.632871Z","shell.execute_reply.started":"2025-07-06T21:01:24.944382Z","shell.execute_reply":"2025-07-06T21:01:40.632326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:40.633517Z","iopub.execute_input":"2025-07-06T21:01:40.633698Z","iopub.status.idle":"2025-07-06T21:01:41.922250Z","shell.execute_reply.started":"2025-07-06T21:01:40.633684Z","shell.execute_reply":"2025-07-06T21:01:41.921658Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Handle Infinite Values","metadata":{}},{"cell_type":"code","source":"train_infs = np.isinf(train.values).sum()\ntest_infs = np.isinf(test.values).sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:41.922942Z","iopub.execute_input":"2025-07-06T21:01:41.923231Z","iopub.status.idle":"2025-07-06T21:01:48.086976Z","shell.execute_reply.started":"2025-07-06T21:01:41.923211Z","shell.execute_reply":"2025-07-06T21:01:48.086363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(f\"Infs in Train: {train_infs}, Infs in Test: {test_infs}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:48.087691Z","iopub.execute_input":"2025-07-06T21:01:48.087911Z","iopub.status.idle":"2025-07-06T21:01:48.092342Z","shell.execute_reply.started":"2025-07-06T21:01:48.087893Z","shell.execute_reply":"2025-07-06T21:01:48.091707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_inf_counts = np.isinf(train).sum(axis=0)\ntest_inf_counts = np.isinf(test).sum(axis=0)\n\ninf_summary = pd.DataFrame({\n    'Train Inf Count': train_inf_counts,\n    'Test Inf Count': test_inf_counts\n})\n\ncolumns_with_inf = inf_summary[(inf_summary['Train Inf Count'] > 0) | (inf_summary['Test Inf Count'] > 0)]\n\nprint(\" Columns containing inf values:\")\nprint(columns_with_inf.sort_values(by='Train Inf Count', ascending=False))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:48.094766Z","iopub.execute_input":"2025-07-06T21:01:48.095125Z","iopub.status.idle":"2025-07-06T21:01:51.951315Z","shell.execute_reply.started":"2025-07-06T21:01:48.095109Z","shell.execute_reply":"2025-07-06T21:01:51.950671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns_with_inf_values = inf_summary[\n    (inf_summary['Train Inf Count'] > 0) | (inf_summary['Test Inf Count'] > 0)\n].index.tolist()\n\n\ntrain.drop(columns=columns_with_inf_values, inplace=True)\ntest.drop(columns=columns_with_inf_values, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:51.951971Z","iopub.execute_input":"2025-07-06T21:01:51.952228Z","iopub.status.idle":"2025-07-06T21:01:55.147725Z","shell.execute_reply.started":"2025-07-06T21:01:51.952202Z","shell.execute_reply":"2025-07-06T21:01:55.147170Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Sample the Data (Optional for Speed)","metadata":{}},{"cell_type":"code","source":"train = train.sample(frac=0.5, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:55.148489Z","iopub.execute_input":"2025-07-06T21:01:55.148685Z","iopub.status.idle":"2025-07-06T21:01:57.377069Z","shell.execute_reply.started":"2025-07-06T21:01:55.148670Z","shell.execute_reply":"2025-07-06T21:01:57.376459Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Visualize the Target Distribution ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 5))\nsns.histplot(train['label'], bins=100, kde=True, color='skyblue')\nplt.title('Distribution of Target Label')\nplt.xlabel('Label')\nplt.ylabel('Frequency')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:57.377758Z","iopub.execute_input":"2025-07-06T21:01:57.377981Z","iopub.status.idle":"2025-07-06T21:01:58.778388Z","shell.execute_reply.started":"2025-07-06T21:01:57.377965Z","shell.execute_reply":"2025-07-06T21:01:58.777663Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Train/Validation Split & Feature Selection","metadata":{}},{"cell_type":"code","source":"train_df, valid_df = train_test_split(train, test_size=0.2, random_state=42)\n\nX_train = train_df.drop(columns=['label'])\ny_train = train_df['label']\n\nX_valid = valid_df.drop(columns=['label'])\ny_valid = valid_df['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:01:58.779043Z","iopub.execute_input":"2025-07-06T21:01:58.779303Z","iopub.status.idle":"2025-07-06T21:02:02.455657Z","shell.execute_reply.started":"2025-07-06T21:01:58.779285Z","shell.execute_reply":"2025-07-06T21:02:02.455134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selector = SelectKBest(score_func=f_regression, k=100)\nX_train = selector.fit_transform(X_train, y_train)\nX_valid = selector.transform(X_valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:02:02.456383Z","iopub.execute_input":"2025-07-06T21:02:02.456575Z","iopub.status.idle":"2025-07-06T21:02:04.251171Z","shell.execute_reply.started":"2025-07-06T21:02:02.456557Z","shell.execute_reply":"2025-07-06T21:02:04.250571Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Train XGBoost Model","metadata":{}},{"cell_type":"code","source":"dtrain = xgb.DMatrix(X_train, label=y_train)\ndvalid = xgb.DMatrix(X_valid, label=y_valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:02:04.251863Z","iopub.execute_input":"2025-07-06T21:02:04.252075Z","iopub.status.idle":"2025-07-06T21:02:04.679703Z","shell.execute_reply.started":"2025-07-06T21:02:04.252060Z","shell.execute_reply":"2025-07-06T21:02:04.679179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n    'objective': 'reg:squarederror',\n    'learning_rate': 0.01,\n    'max_depth': 6,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'tree_method': 'gpu_hist',  \n    'eval_metric': 'rmse',\n    'seed': 42\n}\n\nevallist = [(dtrain, 'train'), (dvalid, 'valid')]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:02:04.680196Z","iopub.execute_input":"2025-07-06T21:02:04.680371Z","iopub.status.idle":"2025-07-06T21:02:04.685588Z","shell.execute_reply.started":"2025-07-06T21:02:04.680356Z","shell.execute_reply":"2025-07-06T21:02:04.685136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = xgb.train(\n    params,\n    dtrain,\n    num_boost_round=10000,\n    evals=evallist,\n    early_stopping_rounds=30,\n    verbose_eval=100\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:02:04.685986Z","iopub.execute_input":"2025-07-06T21:02:04.686280Z","iopub.status.idle":"2025-07-06T21:03:28.810406Z","shell.execute_reply.started":"2025-07-06T21:02:04.686263Z","shell.execute_reply":"2025-07-06T21:03:28.809852Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Train LightGBM Model","metadata":{}},{"cell_type":"code","source":"lgb_train = lgb.Dataset(X_train, label=y_train)\nlgb_valid = lgb.Dataset(X_valid, label=y_valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:03:28.811133Z","iopub.execute_input":"2025-07-06T21:03:28.811344Z","iopub.status.idle":"2025-07-06T21:03:28.815440Z","shell.execute_reply.started":"2025-07-06T21:03:28.811328Z","shell.execute_reply":"2025-07-06T21:03:28.814767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_params = {\n    'objective': 'regression',\n    'metric': 'rmse',\n    'learning_rate': 0.01,\n    'num_leaves': 31,\n    'verbose': -1,\n    'seed': 42\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:03:28.816204Z","iopub.execute_input":"2025-07-06T21:03:28.816749Z","iopub.status.idle":"2025-07-06T21:03:28.832384Z","shell.execute_reply.started":"2025-07-06T21:03:28.816727Z","shell.execute_reply":"2025-07-06T21:03:28.831767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgb_model = lgb.train(\n    lgb_params,\n    lgb_train,\n    valid_sets=[lgb_train, lgb_valid],\n    valid_names=['train', 'valid'],\n    num_boost_round=10000,\n    callbacks=[\n        lgb.early_stopping(stopping_rounds=30),\n        lgb.log_evaluation(period=100)  # بدل verbose_eval\n    ]\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:03:28.833073Z","iopub.execute_input":"2025-07-06T21:03:28.833281Z","iopub.status.idle":"2025-07-06T21:08:18.308457Z","shell.execute_reply.started":"2025-07-06T21:03:28.833258Z","shell.execute_reply":"2025-07-06T21:08:18.307659Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"##  Compare Models (Time & RMSE)","metadata":{}},{"cell_type":"code","source":"def plot_rmse_and_time(results):\n    df = pd.DataFrame(results)\n\n    # Training Time\n    plt.figure(figsize=(10, 5))\n    sns.barplot(x='Model', y='Train Time (s)', data=df, palette='Blues_d')\n    plt.title(\" Training Time Comparison\")\n    plt.ylabel(\"Seconds\")\n    plt.grid(True)\n    plt.show()\n\n    # RMSE\n    plt.figure(figsize=(10, 5))\n    sns.barplot(x='Model', y='RMSE', data=df, palette='Reds_d')\n    plt.title(\"RMSE Comparison on Validation Set\")\n    plt.ylabel(\"RMSE\")\n    plt.grid(True)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:08:18.309381Z","iopub.execute_input":"2025-07-06T21:08:18.309593Z","iopub.status.idle":"2025-07-06T21:08:18.314782Z","shell.execute_reply.started":"2025-07-06T21:08:18.309578Z","shell.execute_reply":"2025-07-06T21:08:18.314146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Training XGBoost...\")\nstart_xgb = time.time()\n\ndtrain = xgb.DMatrix(X_train, label=y_train)\ndvalid = xgb.DMatrix(X_valid, label=y_valid)\n\nxgb_results = {}\nxgb_model = xgb.train(\n    params,\n    dtrain,\n    num_boost_round=10000,\n    evals=[(dtrain, 'train'), (dvalid, 'valid')],\n    early_stopping_rounds=30,\n    evals_result=xgb_results,\n    verbose_eval=False\n)\n\nend_xgb = time.time()\nxgb_time = end_xgb - start_xgb\n\nxgb_preds = xgb_model.predict(dvalid)\nxgb_rmse = mean_squared_error(y_valid, xgb_preds, squared=False)\n\n# -------------------- LightGBM ------------------------\nprint(\" Training LightGBM...\")\nstart_lgb = time.time()\n\nlgb_train_data = lgb.Dataset(X_train, label=y_train)\nlgb_valid_data = lgb.Dataset(X_valid, label=y_valid)\n\nlgb_results = {}\nlgb_model = lgb.train(\n    lgb_params,\n    lgb_train_data,\n    valid_sets=[lgb_train_data, lgb_valid_data],\n    valid_names=['train', 'valid'],\n    num_boost_round=10000,\n    callbacks=[\n        lgb.early_stopping(stopping_rounds=30),\n        lgb.log_evaluation(period=0),\n        lgb.record_evaluation(lgb_results)\n    ]\n)\n\nend_lgb = time.time()\nlgb_time = end_lgb - start_lgb\n\nlgb_preds = lgb_model.predict(X_valid)\nlgb_rmse = mean_squared_error(y_valid, lgb_preds, squared=False)\n\n# -------------------- Results Table + Plot ------------------------\nresults = [\n    {'Model': 'XGBoost', 'Train Time (s)': xgb_time, 'RMSE': xgb_rmse},\n    {'Model': 'LightGBM', 'Train Time (s)': lgb_time, 'RMSE': lgb_rmse},\n]\n\nplot_rmse_and_time(results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:08:18.315456Z","iopub.execute_input":"2025-07-06T21:08:18.315641Z","iopub.status.idle":"2025-07-06T21:14:45.374937Z","shell.execute_reply.started":"2025-07-06T21:08:18.315626Z","shell.execute_reply":"2025-07-06T21:14:45.374215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\nplt.plot(xgb_results['train']['rmse'], label='XGBoost Train RMSE')\nplt.plot(xgb_results['valid']['rmse'], label='XGBoost Valid RMSE')\nplt.plot(lgb_results['train']['rmse'], label='LightGBM Train RMSE')\nplt.plot(lgb_results['valid']['rmse'], label='LightGBM Valid RMSE')\nplt.title(\"📊 RMSE Over Boosting Rounds\")\nplt.xlabel(\"Boosting Round\")\nplt.ylabel(\"RMSE\")\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:14:45.375807Z","iopub.execute_input":"2025-07-06T21:14:45.376090Z","iopub.status.idle":"2025-07-06T21:14:45.641986Z","shell.execute_reply.started":"2025-07-06T21:14:45.376066Z","shell.execute_reply":"2025-07-06T21:14:45.641262Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Importances\n","metadata":{}},{"cell_type":"code","source":"# XGBoost Importance\nxgb.plot_importance(xgb_model, max_num_features=20, importance_type='gain', title='XGBoost Feature Importance (Top 20)', height=0.5)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:14:45.642829Z","iopub.execute_input":"2025-07-06T21:14:45.643129Z","iopub.status.idle":"2025-07-06T21:14:45.952935Z","shell.execute_reply.started":"2025-07-06T21:14:45.643105Z","shell.execute_reply":"2025-07-06T21:14:45.952167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# LightGBM Importance\nlgb.plot_importance(lgb_model, max_num_features=20, importance_type='gain', title='LightGBM Feature Importance (Top 20)', height=0.5)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:14:45.953696Z","iopub.execute_input":"2025-07-06T21:14:45.953970Z","iopub.status.idle":"2025-07-06T21:14:46.262699Z","shell.execute_reply.started":"2025-07-06T21:14:45.953934Z","shell.execute_reply":"2025-07-06T21:14:46.261927Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" ## Submission","metadata":{}},{"cell_type":"code","source":"\nSubmission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:20:38.068892Z","iopub.execute_input":"2025-07-06T21:20:38.069442Z","iopub.status.idle":"2025-07-06T21:20:38.339030Z","shell.execute_reply.started":"2025-07-06T21:20:38.069421Z","shell.execute_reply":"2025-07-06T21:20:38.338457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ncolumns_to_drop = [col for col in columns_with_inf_values if col in test.columns]\n\ntest.drop(columns=columns_to_drop, inplace=True)\n\nX_test = test.drop(columns=['label'], errors='ignore')\n\nX_test_selected = selector.transform(X_test)\ndtest = xgb.DMatrix(X_test_selected)\npredictions = model.predict(dtest)\n\nsubmission = pd.DataFrame({\n    'ID': test.index + 1,\n    'prediction': predictions\n})\n\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T21:31:43.485220Z","iopub.execute_input":"2025-07-06T21:31:43.485776Z","iopub.status.idle":"2025-07-06T21:32:04.925069Z","shell.execute_reply.started":"2025-07-06T21:31:43.485756Z","shell.execute_reply":"2025-07-06T21:32:04.924521Z"}},"outputs":[],"execution_count":null}]}