{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport gc\nfrom sklearn.linear_model import LinearRegression\nimport numpy as np\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error , r2_score\nfrom scipy.stats import pearsonr\nfrom IPython.display import display\nimport math\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nfrom tqdm import tqdm\nimport xgboost as xgb\nfrom catboost import CatBoostRegressor\nfrom lightgbm import LGBMRegressor\nfrom sklearn.ensemble import StackingRegressor\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"12711ee8-b91d-4c9b-ab67-935c888b7f2d","_cell_guid":"79d676b5-d7f2-4295-8779-a19fcb983bdb","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:19:19.522162Z","iopub.execute_input":"2025-07-24T20:19:19.522868Z","iopub.status.idle":"2025-07-24T20:19:26.425110Z","shell.execute_reply.started":"2025-07-24T20:19:19.522841Z","shell.execute_reply":"2025-07-24T20:19:26.424538Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet', engine='pyarrow')","metadata":{"_uuid":"ac4448cc-91db-492f-85df-71932c2d52ff","_cell_guid":"d7e62877-c5d1-47e2-9b23-04c0759c5409","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:19:26.426168Z","iopub.execute_input":"2025-07-24T20:19:26.426687Z","iopub.status.idle":"2025-07-24T20:19:46.622432Z","shell.execute_reply.started":"2025-07-24T20:19:26.426668Z","shell.execute_reply":"2025-07-24T20:19:46.621794Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(5)","metadata":{"_uuid":"f6041959-ece3-48da-85fe-8a521f2ecdf4","_cell_guid":"871c72ed-795c-44d2-bf6e-376aa710e58f","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:19:46.623184Z","iopub.execute_input":"2025-07-24T20:19:46.623454Z","iopub.status.idle":"2025-07-24T20:19:46.659040Z","shell.execute_reply.started":"2025-07-24T20:19:46.623430Z","shell.execute_reply":"2025-07-24T20:19:46.658339Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = df.iloc[:500000]\ntest_df = df.iloc[500000:]\ntrain_df = train_df.astype('float32')\ntest_df = test_df.astype('float32')","metadata":{"_uuid":"a44a6e0d-7aed-4897-8201-abf3d503af05","_cell_guid":"572b933d-cd0c-4ac4-8ed5-fc4a1975b4a3","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:19:46.659782Z","iopub.execute_input":"2025-07-24T20:19:46.660089Z","iopub.status.idle":"2025-07-24T20:19:47.310135Z","shell.execute_reply.started":"2025-07-24T20:19:46.660068Z","shell.execute_reply":"2025-07-24T20:19:47.309540Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.replace([np.inf, -np.inf], 0, inplace=True)\n\nx_train = train_df.drop(columns = ['label'])\ny_train = train_df['label']\nx_test = test_df.drop(columns = ['label'])\ny_test = test_df['label']","metadata":{"_uuid":"797f59ba-6dc8-4803-913b-80827d8bc09e","_cell_guid":"3adf48c5-99e9-4ca7-b149-538bc5545c33","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:19:47.312383Z","iopub.execute_input":"2025-07-24T20:19:47.312764Z","iopub.status.idle":"2025-07-24T20:19:50.025885Z","shell.execute_reply.started":"2025-07-24T20:19:47.312744Z","shell.execute_reply":"2025-07-24T20:19:50.025283Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def find_cols_with_zero_variance(x_train):\n    zero_var_cols = x_train.columns[x_train.nunique() <= 1].tolist()\n    print(\"Columns with 0 variance:\", zero_var_cols)\n    print(f\"{len(zero_var_cols)} col skipped\")\n    return zero_var_cols\n    \ndef find_cols_with_low_correlation(x_train, y_train, threshold=0.01):\n    correlations = x_train.corrwith(y_train)\n    low_corr_cols = correlations[correlations.abs() < threshold].index.tolist()\n    print(\"Columns with low correlation:\", low_corr_cols)\n    print(f\"{len(low_corr_cols)} features dropped with correlation below {threshold}.\")\n    return low_corr_cols\n\ndef find_top_features_with_xgboost(x_train, x_test, y_train, top_n=50):\n    model = xgb.XGBRegressor(\n        objective='reg:squarederror',\n        n_estimators=100,\n        learning_rate=0.1,\n        max_depth=6,\n        random_state=42,\n        verbosity=0\n    )\n    model.fit(x_train, y_train)\n\n    importances = model.feature_importances_\n    feature_names = x_train.columns\n\n    importance_df = pd.DataFrame({\n        'feature': feature_names,\n        'importance': importances\n    }).sort_values(by='importance', ascending=False)\n\n    top_features = importance_df.head(top_n)['feature'].tolist()\n    dropped_features = importance_df.tail(len(importance_df) - top_n)['feature'].tolist()\n\n    print(f\"Selected top {top_n} features based on XGBoost importance.\")\n    print(f\"Dropped features: {dropped_features}\")\n\n    return top_features\n\ndef select_top_features_with_xgboost(data, top_features):\n    return data[top_features]\n\ndef find_highly_correlated_features(x_train, threshold=0.98):\n    corr_matrix = x_train.corr().abs()\n    upper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))\n    high_corr_cols = [column for column in upper.columns if any(upper[column] > threshold)]\n    print(f\"Dropping {len(high_corr_cols)} highly correlated features (threshold > {threshold})\")\n    print(f\"Columns dropped: {high_corr_cols}\")\n    return high_corr_cols\n\ndef drop_features(data, to_drop):\n    return data.drop(columns=to_drop)\n\ndef process(x_train, x_test, y_train):\n    zero_var_cols = find_cols_with_zero_variance(x_train)\n    x_train = drop_features(x_train, zero_var_cols)\n    x_test = drop_features(x_test, zero_var_cols)\n    \n    low_corr_cols = find_cols_with_low_correlation(x_train, y_train, threshold=0.01) \n    x_train = drop_features(x_train, low_corr_cols)\n    x_test = drop_features(x_test, low_corr_cols)\n    \n    selected_features = find_top_features_with_xgboost(x_train, x_test, y_train, top_n=300)\n    x_train = select_top_features_with_xgboost(x_train, selected_features)\n    x_test = select_top_features_with_xgboost(x_test, selected_features)\n\n    high_corr_cols = find_highly_correlated_features(x_train, 0.98)\n    x_train = drop_features(x_train, high_corr_cols)\n    x_test = drop_features(x_test, high_corr_cols)\n    \n    return x_train, x_test, y_train\n\ndef train_fit_xgboost(x_train , y_train):\n    XGBR = xgb.XGBRegressor(\n            objective='reg:squarederror',\n            n_estimators=500,\n            learning_rate=0.1,\n            random_state=42,\n    )\n    XGBR.fit(x_train , y_train)\n    return XGBR\n\ndef predict_res(model, x_test):\n    y_pred = model.predict(x_test)\n    return y_pred\n\n\"\"\"xgb_model = xgb.XGBRegressor(\n    objective='reg:squarederror',\n    n_estimators=500,\n    learning_rate=0.1,\n    #max_depth=6,\n    random_state=42,\n)\n\n# LightGBM\nlgb_model = LGBMRegressor(\n    n_estimators=300,\n    learning_rate=0.1,\n    max_depth=6,\n    random_state=42\n)\n\n# CatBoost\ncat_model = CatBoostRegressor(\n    iterations=300,\n    learning_rate=0.1,\n    depth=6,\n    verbose=0,\n    random_state=42\n)\n\n# Ensemble: Stacking\nstacking_model = StackingRegressor(\n    estimators=[\n        ('xgb', xgb_model),\n        ('lgb', lgb_model),\n        ('cat', cat_model)\n    ],\n    final_estimator=Ridge(),\n    passthrough=True,\n    cv=3\n)\n\nprint(\"Training Stacking Model...\")\nstacking_model.fit(x_train, y_train)\"\"\"\n\n\ndef find_errors_and_plot(y_test, y_pred, model_name):\n    mae = mean_absolute_error(y_test, y_pred)\n    mse = mean_squared_error(y_test, y_pred)\n    rmse = np.sqrt(mse)\n    r2 = r2_score(y_test, y_pred)\n    corr_coef, p_value = pearsonr(y_test, y_pred)\n    \n    results = pd.DataFrame({\n        'Model': [f'{model_name}'],\n        'MAE': [mae],\n        'MSE': [mse],\n        'RMSE': [rmse],\n        'R2': [r2],\n        'Pearson Correlation Coefficient':[corr_coef],\n        'P value':[p_value],\n    })\n    display(results)\n    \n    np.random.seed(42)\n    sample_indices = np.random.choice(len(y_test), size=100, replace=False)\n    \n    y_test_sample = y_test.iloc[sample_indices].reset_index(drop=True)\n    y_pred_sample = pd.Series(y_pred[sample_indices])\n    \n    plt.figure(figsize=(12, 6))\n    plt.plot(y_test_sample, label='Actual', marker='o')\n    plt.plot(y_pred_sample, label='Predicted', marker='x')\n    plt.title(\"Actual vs Predicted (100 Random Samples)\")\n    plt.xlabel(\"Sample Index\")\n    plt.ylabel(\"Target Value\")\n    plt.legend()\n    plt.tight_layout()\n    plt.show()","metadata":{"_uuid":"7a1aaaf1-102b-43fb-af8b-e61bc0efc660","_cell_guid":"023b9512-5465-4aaf-9699-c0f62620c075","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:19:50.026614Z","iopub.execute_input":"2025-07-24T20:19:50.026823Z","iopub.status.idle":"2025-07-24T20:19:50.041202Z","shell.execute_reply.started":"2025-07-24T20:19:50.026806Z","shell.execute_reply":"2025-07-24T20:19:50.040557Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train, x_test, y_train = process(x_train, x_test, y_train)","metadata":{"_uuid":"ad5b4ecc-f375-4f80-a930-9464343ba480","_cell_guid":"1d009e59-f837-4035-a65c-2b04af867798","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:19:50.041878Z","iopub.execute_input":"2025-07-24T20:19:50.042127Z","iopub.status.idle":"2025-07-24T20:22:58.850318Z","shell.execute_reply.started":"2025-07-24T20:19:50.042098Z","shell.execute_reply":"2025-07-24T20:22:58.849452Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_feature_vs_label(x, y, features):\n    for feature in features:\n        plt.figure(figsize=(6, 4))\n        sns.scatterplot(x=x[feature], y=y, s=10, alpha=0.5)\n        plt.xlabel(feature)\n        plt.ylabel('label')\n        plt.title(f\"{feature} vs label\")\n        plt.tight_layout()\n        plt.show()","metadata":{"_uuid":"6f84ade0-6e57-4ba7-92b7-8104c1d77169","_cell_guid":"a4561b3b-ab1a-4011-a10d-7a1dae956dbe","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:22:58.851171Z","iopub.execute_input":"2025-07-24T20:22:58.851443Z","iopub.status.idle":"2025-07-24T20:22:58.855967Z","shell.execute_reply.started":"2025-07-24T20:22:58.851413Z","shell.execute_reply":"2025-07-24T20:22:58.855438Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#correlation_series = x_train.corrwith(y_train).abs().sort_values(ascending=False)\n#top_corr_features = correlation_series.head(300).index.tolist()\n#plot_feature_vs_label(x_train, y_train, features=top_corr_features)","metadata":{"_uuid":"657e6ff8-3ec1-46f0-9f7e-bba7b2bb0d1e","_cell_guid":"07c33da3-f80b-40fe-b3be-47ecc8472f0b","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:22:58.856616Z","iopub.execute_input":"2025-07-24T20:22:58.856834Z","iopub.status.idle":"2025-07-24T20:22:58.880373Z","shell.execute_reply.started":"2025-07-24T20:22:58.856808Z","shell.execute_reply":"2025-07-24T20:22:58.879714Z"},"scrolled":true,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(x_train.columns))\n\nXGBR = train_fit_xgboost(x_train , y_train)\n\n#y_pred = predict_res(stacking_model, x_test)\ny_pred = predict_res(XGBR, x_test)\n\nfind_errors_and_plot(y_test, y_pred, 'XGB')","metadata":{"_uuid":"e4b83b3f-18da-41c3-b1b9-c1843980b3ee","_cell_guid":"430fa44b-7af5-4f5c-b8bd-1d6c13efc61d","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:22:58.881189Z","iopub.execute_input":"2025-07-24T20:22:58.881391Z","iopub.status.idle":"2025-07-24T20:24:43.502182Z","shell.execute_reply.started":"2025-07-24T20:22:58.881376Z","shell.execute_reply":"2025-07-24T20:24:43.501492Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install shap\nimport shap","metadata":{"_uuid":"b0c77d58-4248-4fde-9a96-bf009542147e","_cell_guid":"f6c6722a-5c98-4206-a8fa-3c16278a3a94","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-24T20:24:43.502966Z","iopub.execute_input":"2025-07-24T20:24:43.503160Z","iopub.status.idle":"2025-07-24T20:24:54.283618Z","shell.execute_reply.started":"2025-07-24T20:24:43.503145Z","shell.execute_reply":"2025-07-24T20:24:54.282928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def shap_analysis(x_train, model, max_display=20, sample_size=1000):\n    sampled_x = x_train.sample(sample_size, random_state=42)\n    explainer = shap.Explainer(model)\n    shap_values = explainer(sampled_x)\n\n    print(\"Generating SHAP summary plot...\")\n    shap.summary_plot(shap_values, sampled_x, plot_type=\"bar\", max_display=max_display)\n    shap.summary_plot(shap_values, sampled_x, max_display=max_display)\n    shap.plots.waterfall(shap_values[100])\n\nshap_analysis(x_train, XGBR, max_display=50, sample_size=1000)","metadata":{"_uuid":"c42cbba4-faf1-404f-b679-5f700a9862f5","_cell_guid":"62183882-c43b-445b-8ef0-0f64df0776fe","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:24:54.284484Z","iopub.execute_input":"2025-07-24T20:24:54.285030Z","iopub.status.idle":"2025-07-24T20:25:02.314684Z","shell.execute_reply.started":"2025-07-24T20:24:54.285007Z","shell.execute_reply":"2025-07-24T20:25:02.313896Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del x_train\ndel y_train\ndel x_test\ndel y_test\ndel df\ngc.collect()\n%system free -m","metadata":{"_uuid":"7739b4a7-fe40-4513-addf-87fd8d566fae","_cell_guid":"07da6ef3-86fd-4371-b5f7-e1126b2b0027","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:25:02.315571Z","iopub.execute_input":"2025-07-24T20:25:02.315813Z","iopub.status.idle":"2025-07-24T20:25:02.561134Z","shell.execute_reply.started":"2025-07-24T20:25:02.315794Z","shell.execute_reply":"2025-07-24T20:25:02.560406Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## SUBMISSON\ntrain_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet', engine='pyarrow')\ntest_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet', engine='pyarrow')\n\ntrain_df = train_df.astype('float32')\ntest_df = test_df.astype('float32')\n\ntrain_df.replace([np.inf, -np.inf], 0, inplace=True)\ntest_df.replace([np.inf, -np.inf], 0, inplace=True)\n\nx_train = train_df.drop(columns = ['label'])\ny_train = train_df['label']\nx_test = test_df.drop(columns = ['label'])\ny_test = test_df['label']\n\ndel train_df\ndel test_df\ngc.collect()\n\nx_train, x_test, y_train = process(x_train, x_test, y_train)\n\n# Train\nXGBR = train_fit_xgboost(x_train , y_train)\n\n# Test\n#y_pred = predict_res(stacking_model, x_test)\ny_pred = predict_res(XGBR, x_test)\ny_pred","metadata":{"_uuid":"63772070-3edd-4f50-a00e-34b5c3a9025a","_cell_guid":"1d6925de-9200-46e9-87df-f314854f9418","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:25:02.563141Z","iopub.execute_input":"2025-07-24T20:25:02.563383Z","iopub.status.idle":"2025-07-24T20:31:07.046296Z","shell.execute_reply.started":"2025-07-24T20:25:02.563366Z","shell.execute_reply":"2025-07-24T20:31:07.045583Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\nsample_submission['prediction'] = y_pred\nsample_submission.to_csv('submission.csv',index = False )","metadata":{"_uuid":"b177c178-2b86-4c62-9d9b-759fafd88f6b","_cell_guid":"0479daf9-5e0e-4a44-8db6-f13b15ff0c0a","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-07-24T20:31:07.047054Z","iopub.execute_input":"2025-07-24T20:31:07.047291Z","iopub.status.idle":"2025-07-24T20:31:08.080927Z","shell.execute_reply.started":"2025-07-24T20:31:07.047252Z","shell.execute_reply":"2025-07-24T20:31:08.080315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission","metadata":{"_uuid":"9d91eb9f-3095-426e-baa1-3cf65a465ec2","_cell_guid":"81d1fff0-7d10-4c7f-9d57-6868747fbf3a","trusted":true,"collapsed":false,"execution":{"iopub.status.busy":"2025-07-24T20:31:08.081677Z","iopub.execute_input":"2025-07-24T20:31:08.081946Z","iopub.status.idle":"2025-07-24T20:31:08.094720Z","shell.execute_reply.started":"2025-07-24T20:31:08.081920Z","shell.execute_reply":"2025-07-24T20:31:08.094156Z"},"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}