{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":12503273,"sourceType":"datasetVersion","datasetId":7870836},{"sourceId":12462463,"sourceType":"datasetVersion","datasetId":7861436}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-21T03:50:26.177323Z","iopub.execute_input":"2025-07-21T03:50:26.177743Z","iopub.status.idle":"2025-07-21T03:50:26.194927Z","shell.execute_reply.started":"2025-07-21T03:50:26.177715Z","shell.execute_reply":"2025-07-21T03:50:26.193704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# DRW - Crypto Market Prediction | Mohamed Zakaria\n\n# Imports\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import KFold\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_squared_error\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom tensorflow.keras.callbacks import EarlyStopping\n\n# Load Data\ntrain_df = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/train.parquet\")\ntest_df = pd.read_parquet(\"/kaggle/input/drw-crypto-market-prediction/test.parquet\")\n\n# Basic EDA\nprint(\"\\n Data Overview:\\n\", train_df.head())\nprint(\"\\n Missing Values:\\n\", train_df.isnull().sum().sort_values(ascending=False))\nprint(\"\\n Shape:\", train_df.shape)\n\n# EDA Visualization\nplt.figure(figsize=(12, 6))\nsns.heatmap(train_df.corr(numeric_only=True), cmap='coolwarm', annot=False)\nplt.title(\"Feature Correlation Heatmap\")\nplt.show()\n\n# Lag Feature Engineering (using index order)\ndef create_lag_features(df, lags=[1, 2]):\n    df = df.copy()\n    FEATURES = [col for col in df.columns if col != \"label\" and col != \"row_id\"]\n\n    lagged_features = []\n    for lag in lags:\n        lagged = df[FEATURES].shift(lag)\n        lagged.columns = [f\"{col}_lag_{lag}\" for col in FEATURES]\n        lagged_features.append(lagged)\n\n    df_lagged = pd.concat([df] + lagged_features, axis=1)\n    return df_lagged\n\ntrain = create_lag_features(train_df)\ntrain.dropna(inplace=True)\n\n# 🫒 Features & Target\nFEATURES = [col for col in train.columns if col != \"label\" and col != \"row_id\"]\nX = train[FEATURES].values\ny = train[\"label\"].values\n\n# Scale Features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# Neural Network Model\ndef create_model(input_dim):\n    model = Sequential()\n    model.add(Dense(256, activation='relu', input_dim=input_dim))\n    model.add(Dropout(0.4))\n    model.add(Dense(128, activation='relu'))\n    model.add(Dropout(0.3))\n    model.add(Dense(64, activation='relu'))\n    model.add(Dropout(0.2))\n    model.add(Dense(1))\n    model.compile(optimizer='adam', loss='mse')\n    return model\n\n# Cross Validation & Training\nkf = KFold(n_splits=5, shuffle=True, random_state=42)\noof_preds = np.zeros(len(X))\nmodels = []\nfold_scores = []\n\nfor fold, (train_idx, val_idx) in enumerate(kf.split(X_scaled)):\n    print(f\"\\n🫒 Fold {fold+1}\")\n    X_train, X_val = X_scaled[train_idx], X_scaled[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n\n    model = create_model(X_train.shape[1])\n    es = EarlyStopping(patience=10, restore_best_weights=True)\n    model.fit(X_train, y_train, validation_data=(X_val, y_val),\n              epochs=100, batch_size=128, callbacks=[es], verbose=0)\n\n    val_preds = model.predict(X_val).flatten()\n    oof_preds[val_idx] = val_preds\n    rmse = np.sqrt(mean_squared_error(y_val, val_preds))\n    fold_scores.append(rmse)\n    print(\"RMSE:\", rmse)\n    models.append(model)\n\n# Overall CV Score\noverall_rmse = np.sqrt(mean_squared_error(y, oof_preds))\nprint(\"\\n Overall RMSE:\", overall_rmse)\n\n# Fold Results Table\nfold_results = pd.DataFrame({\n    \"Fold\": [f\"Fold {i+1}\" for i in range(len(fold_scores))],\n    \"RMSE\": fold_scores\n})\nfold_results.loc[\"Mean\"] = [\"Average\", fold_results[\"RMSE\"].mean()]\nprint(\"\\n Fold Performance:\")\nprint(fold_results)\n\n# Save Fold Table\nfold_results.to_csv(\"fold_scores.csv\", index=False)\n\n# Plot RMSE per Fold\nplt.figure(figsize=(8, 5))\nsns.barplot(x=\"Fold\", y=\"RMSE\", data=fold_results[:-1], palette=\"viridis\")\nplt.axhline(fold_results.loc[\"Mean\", \"RMSE\"], ls='--', color='red', label=\"Mean RMSE\")\nplt.title(\"Cross-Validation RMSE per Fold\")\nplt.ylabel(\"RMSE\")\nplt.legend()\nplt.tight_layout()\nplt.savefig(\"rmse_per_fold.png\")\nplt.show()\n\n# Compare Real vs Predicted\nplt.figure(figsize=(8, 6))\nsns.kdeplot(y, label=\"True Label\", linewidth=2)\nsns.kdeplot(oof_preds, label=\"OOF Preds\", linewidth=2)\nplt.legend()\nplt.title(\"True vs OOF Prediction Distribution\")\nplt.savefig(\"true_vs_pred_distribution.png\")\nplt.show()\n\n# Save Real vs Pred CSV\ncomparison_df = pd.DataFrame({\"True_Label\": y, \"OOF_Pred\": oof_preds})\ncomparison_df.to_csv(\"real_vs_pred.csv\", index=False)\n\n# Submission Prediction (blended ensemble)\ntest = create_lag_features(test_df)\ntest.fillna(method='bfill', inplace=True)\nX_test = test[FEATURES].values\nX_test_scaled = scaler.transform(X_test)\n\n# Average blending\nfinal_preds = np.mean([model.predict(X_test_scaled).flatten() for model in models], axis=0)\n\n# Sample submission\nsubmission = pd.DataFrame({\n    \"row_id\": test_df[\"row_id\"],\n    \"label\": final_preds\n})\n\n# Preview Table\nprint(\"\\n Submission Preview:\")\nprint(submission.head())\n\nsubmission.to_csv(\"submission.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-21T04:10:06.714396Z","iopub.execute_input":"2025-07-21T04:10:06.714853Z","execution_failed":"2025-07-21T04:27:23.673Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DRW - Crypto Market Prediction | Neural Network Blending\n\nWelcome to this end-to-end notebook for the **DRW - Crypto Market Prediction** competition on Kaggle, built by **Mohamed Zakaria**.\n\nThis notebook showcases a full pipeline for time-series tabular prediction using:\n\n-  **Lag-based feature engineering**\n-  **Deep Neural Network modeling**\n-  **KFold Cross-Validation with RMSE evaluation**\n-  **Visual comparison: Real vs. Predicted distributions**\n-  **Ensemble predictions via NN blending**\n-  **Exported CSVs for fold results, predictions & submission**\n\n---\n\n### Objectives:\n\n1. Build a robust model to predict the short-term price movement (`label`) using engineered features.\n2. Analyze overfitting/generalization using fold-wise RMSE.\n3. Visualize and compare real vs. predicted distributions.\n4. Save all outputs clearly for further inspection.\n\n---\n\n> 📦 Files saved from this notebook:\n> - `submission.csv`: Final test predictions  \n> - `fold_scores.csv`: RMSE per fold  \n> - `real_vs_pred.csv`: Ground-truth vs. OOF predictions  \n> - `rmse_per_fold.png`: Fold RMSE barplot  \n> - `true_vs_pred_distribution.png`: Density plot of real vs. predicted values\n\nGood luck and may the **crypto signal** be with you  \n","metadata":{}}]}