{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:56:59.878554Z","iopub.execute_input":"2025-07-09T11:56:59.878819Z","iopub.status.idle":"2025-07-09T11:57:01.827659Z","shell.execute_reply.started":"2025-07-09T11:56:59.878789Z","shell.execute_reply":"2025-07-09T11:57:01.826749Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data & Preprocessing\n","metadata":{}},{"cell_type":"code","source":"# Load data\ndf_train = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ndf_test = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:57:01.828659Z","iopub.execute_input":"2025-07-09T11:57:01.829168Z","iopub.status.idle":"2025-07-09T11:57:54.239269Z","shell.execute_reply.started":"2025-07-09T11:57:01.829142Z","shell.execute_reply":"2025-07-09T11:57:54.238509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:57:54.241611Z","iopub.execute_input":"2025-07-09T11:57:54.241926Z","iopub.status.idle":"2025-07-09T11:57:54.287268Z","shell.execute_reply.started":"2025-07-09T11:57:54.241901Z","shell.execute_reply":"2025-07-09T11:57:54.286363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:57:54.288254Z","iopub.execute_input":"2025-07-09T11:57:54.288618Z","iopub.status.idle":"2025-07-09T11:57:54.297388Z","shell.execute_reply.started":"2025-07-09T11:57:54.288591Z","shell.execute_reply":"2025-07-09T11:57:54.296400Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA Pipeline for Train and Test Data\nThis function performs a comprehensive exploratory data analysis (EDA) on df_train and df_test, including missing value checks, summary statistics, data types, correlations, outlier detection, and distribution visualization for numerical and categorical features.","metadata":{}},{"cell_type":"markdown","source":"**Make sure to change the name of the target variable.**","metadata":{}},{"cell_type":"code","source":"target_variable=\"label\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:57:54.298539Z","iopub.execute_input":"2025-07-09T11:57:54.299700Z","iopub.status.idle":"2025-07-09T11:57:54.314776Z","shell.execute_reply.started":"2025-07-09T11:57:54.299663Z","shell.execute_reply":"2025-07-09T11:57:54.313664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n  \ndef eda_pipeline(df_train, df_test):\n    \n    # Display first few rows\n    print(\"\\n--- First few rows of train data ---\")\n    display(df_train.head())\n    \n    print(\"\\n--- First few rows of test data ---\")\n    display(df_test.head())\n    \n    # Dataset info\n    print(\"\\n--- Train Data Info ---\")\n    print(df_train.info())\n    \n    print(\"\\n--- Test Data Info ---\")\n    print(df_test.info())\n    \n    # Missing values\n    print(\"\\n--- Missing Values in Train Data ---\")\n    print(df_train.isnull().sum())\n    \n    print(\"\\n--- Missing Values in Test Data ---\")\n    print(df_test.isnull().sum())\n    \n    print(\"\\n--- Percentage of Missing Values in Train Data ---\")\n    print((df_train.isnull().sum() / len(df_train)) * 100)\n    \n    print(\"\\n--- Percentage of Missing Values in Test Data ---\")\n    print((df_test.isnull().sum() / len(df_test)) * 100)\n    \n    # # Summary statistics\n    # print(\"\\n--- Train Data Summary Statistics ---\")\n    # print(df_train.describe())\n    \n    # print(\"\\n--- Test Data Summary Statistics ---\")\n    # print(df_test.describe())\n    \n    # Identify categorical columns\n    train_cat_columns = [col for col in df_train.columns if df_train[col].dtype == 'O']\n    test_cat_columns = [col for col in df_test.columns if df_test[col].dtype == 'O']\n    \n    print(\"\\n--- Categorical Columns in Train Data ---\")\n    print(train_cat_columns)\n    \n    print(\"\\n--- Unique Values in Categorical Columns (Train) ---\")\n    print(df_train[train_cat_columns].nunique())\n    \n    print(\"\\n--- Categorical Columns in Test Data ---\")\n    print(test_cat_columns)\n    \n    print(\"\\n--- Unique Values in Categorical Columns (Test) ---\")\n    print(df_test[test_cat_columns].nunique())\n    \n    # # Identify numerical columns\n    # train_num_columns = [col for col in df_train.columns if df_train[col].dtype in ['int64', 'float64']]\n    # test_num_columns = [col for col in df_test.columns if df_test[col].dtype in ['int64', 'float64']]\n    \n    # print(\"\\n--- Numerical Columns in Train Data ---\")\n    # print(train_num_columns)\n    \n    # print(\"\\n--- Numerical Columns in Test Data ---\")\n    # print(test_num_columns)\n    \n    # Check for duplicate rows\n    print(\"\\n--- Duplicate Rows in Train Data ---\")\n    print(df_train.duplicated().sum())\n    \n    print(\"\\n--- Duplicate Rows in Test Data ---\")\n    print(df_test.duplicated().sum())\n    \n    # # Correlation matrix (excluding non-numeric columns)\n    # print(\"\\n--- Correlation Matrix ---\")\n    # plt.figure(figsize=(12, 6))\n    # sns.heatmap(df_train[train_num_columns].corr(), annot=True, cmap='coolwarm')\n    # plt.show()\n       \n    # # Correlation with Target Variable\n    # print(\"\\n--- Correlation with Target Variable ---\")\n    # target_corr = df_train[train_num_columns].corr()[target_variable].sort_values(ascending=False)\n    # print(target_corr)\n    \n    # plt.figure(figsize=(12, 6))\n    # sns.barplot(x=target_corr.index, y=target_corr.values, palette='coolwarm')\n    # plt.xticks(rotation=90)\n    # plt.title(f'Feature Correlation with {target_variable}')\n    # plt.show()   \n    \n    # # Distribution plots for numerical features\n    # print(\"\\n--- Distribution of Numerical Features ---\")\n    # df_train[train_num_columns].hist(figsize=(12, 10), bins=30)\n    # plt.show()\n    \n    # # Box plots for outlier detection\n    # print(\"\\n--- Box Plots for Outlier Detection ---\")\n    # for col in train_num_columns:\n    #     plt.figure(figsize=(8, 4))\n    #     sns.boxplot(x=df_train[col])\n    #     plt.title(f'Box plot of {col}')\n    #     plt.show()\n    \n    # # Value counts for categorical features\n    # print(\"\\n--- Value Counts for Categorical Columns ---\")\n    # for col in train_cat_columns:\n    #     print(f\"\\nValue counts for {col}:\")\n    #     print(df_train[col].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:57:54.318087Z","iopub.execute_input":"2025-07-09T11:57:54.318460Z","iopub.status.idle":"2025-07-09T11:57:55.469710Z","shell.execute_reply.started":"2025-07-09T11:57:54.318431Z","shell.execute_reply":"2025-07-09T11:57:55.468804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eda_pipeline(df_train, df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:57:55.470495Z","iopub.execute_input":"2025-07-09T11:57:55.470964Z","iopub.status.idle":"2025-07-09T11:59:33.174743Z","shell.execute_reply.started":"2025-07-09T11:57:55.470940Z","shell.execute_reply":"2025-07-09T11:59:33.173318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Since there are no missing values, duplicate, categorical columns so we dont need to do any data processing.","metadata":{}},{"cell_type":"code","source":"# # Target Distribution Check\n# print(\"\\n--- Distribution of Target Variable for Class Balance Check ---\\n\")\n# df_train[target_variable].value_counts(normalize=True).plot(kind='barh')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:33.176186Z","iopub.execute_input":"2025-07-09T11:59:33.176562Z","iopub.status.idle":"2025-07-09T11:59:33.182528Z","shell.execute_reply.started":"2025-07-09T11:59:33.176528Z","shell.execute_reply":"2025-07-09T11:59:33.181231Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing Pipeline\nThis function preprocesses df_train and df_test by handling missing values (filling categorical with mode and numerical with mean) and encoding categorical variables using LabelEncoder, ensuring consistency between train and test datasets.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\ndef data_preprocessing_pipeline(df_train, df_test, target_column='label'):\n    \"\"\"\n    Preprocess the dataset by handling missing values and encoding categorical variables.\n    Returns processed DataFrames and the label encoder for the target column.\n    \"\"\"\n    # Fill missing values\n    for column in df_train.columns:\n        if df_train[column].dtype == 'object':\n            mode_value = df_train[column].mode()[0]\n            df_train[column].fillna(mode_value, inplace=True)\n        elif df_train[column].dtype in ['int64', 'float64']:\n            mean_value = df_train[column].mean()\n            df_train[column].fillna(mean_value, inplace=True)\n    \n    for column in df_test.columns:\n        if df_test[column].dtype == 'object':\n            mode_value = df_test[column].mode()[0]\n            df_test[column].fillna(mode_value, inplace=True)\n        elif df_test[column].dtype in ['int64', 'float64']:\n            mean_value = df_test[column].mean()\n            df_test[column].fillna(mean_value, inplace=True)\n    \n    # Encode categorical features\n    label_encoders = {}\n    target_encoder = None  # separate encoder for target column\n\n    for column in df_train.columns:\n        if df_train[column].dtype == 'object':\n            le = LabelEncoder()\n            df_train[column] = le.fit_transform(df_train[column].astype(str))\n            label_encoders[column] = le\n\n            if column == target_column:\n                target_encoder = le  # store encoder for target\n\n    for column in df_test.columns:\n        if df_test[column].dtype == 'object':\n            if column in label_encoders:\n                le = label_encoders[column]\n                df_test[column] = df_test[column].apply(\n                    lambda x: le.transform([x])[0] if x in le.classes_ else -1\n                )\n            else:\n                df_test[column] = -1\n\n    return df_train, df_test, target_encoder","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:33.183834Z","iopub.execute_input":"2025-07-09T11:59:33.184716Z","iopub.status.idle":"2025-07-09T11:59:33.387350Z","shell.execute_reply.started":"2025-07-09T11:59:33.184679Z","shell.execute_reply":"2025-07-09T11:59:33.386243Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Since there no missing values, and categorical data.","metadata":{}},{"cell_type":"code","source":"# data_preprocessing_pipeline(df_train, df_test, target_column='label')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:33.388446Z","iopub.execute_input":"2025-07-09T11:59:33.388895Z","iopub.status.idle":"2025-07-09T11:59:33.393940Z","shell.execute_reply.started":"2025-07-09T11:59:33.388858Z","shell.execute_reply":"2025-07-09T11:59:33.392937Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Standardization of Numerical Features\nThis function standardizes all numerical features in df_train and df_test using StandardScaler, ensuring consistency while preserving the target variable in the train dataset.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\ndef standardize_data(df_train, df_test):\n    \"\"\"\n    Standardize all numerical features using StandardScaler,\n    ensuring both train and test have the same columns, while preserving the target variable.\n    \"\"\"\n    # Separate target column from train data\n    target_values = df_train[target_variable]\n    df_train = df_train.drop(columns=[target_variable])\n    \n    # Ensure both datasets have the same feature columns\n    common_columns = df_train.columns.intersection(df_test.columns)\n    df_train = df_train[common_columns]\n    df_test = df_test[common_columns]\n    \n    # Initialize StandardScaler\n    scaler = StandardScaler()\n    \n    # Fit on train data and transform both train and test data\n    df_train_scaled = pd.DataFrame(scaler.fit_transform(df_train), columns=common_columns)\n    df_test_scaled = pd.DataFrame(scaler.transform(df_test), columns=common_columns)\n    \n    # Reattach the target column to the scaled train data\n    df_train_scaled[target_variable] = target_values.reset_index(drop=True)\n    \n    return df_train_scaled, df_test_scaled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:33.395186Z","iopub.execute_input":"2025-07-09T11:59:33.395560Z","iopub.status.idle":"2025-07-09T11:59:33.412730Z","shell.execute_reply.started":"2025-07-09T11:59:33.395529Z","shell.execute_reply":"2025-07-09T11:59:33.411492Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_train_scaled, df_test_scaled = standardize_data(df_train, df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:33.413942Z","iopub.execute_input":"2025-07-09T11:59:33.414254Z","iopub.status.idle":"2025-07-09T11:59:33.437308Z","shell.execute_reply.started":"2025-07-09T11:59:33.414230Z","shell.execute_reply":"2025-07-09T11:59:33.436165Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature-Target Separation\nThis code separates the features (X) and the target variable (y) from df_train, where X contains all columns except the target, and y stores the target variable values.","metadata":{}},{"cell_type":"code","source":"X = df_train.drop(columns=[target_variable])\ny = df_train[target_variable]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:33.438614Z","iopub.execute_input":"2025-07-09T11:59:33.439023Z","iopub.status.idle":"2025-07-09T11:59:37.166150Z","shell.execute_reply.started":"2025-07-09T11:59:33.439000Z","shell.execute_reply":"2025-07-09T11:59:37.165089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:37.167371Z","iopub.execute_input":"2025-07-09T11:59:37.167689Z","iopub.status.idle":"2025-07-09T11:59:37.191759Z","shell.execute_reply.started":"2025-07-09T11:59:37.167650Z","shell.execute_reply":"2025-07-09T11:59:37.190866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:37.192703Z","iopub.execute_input":"2025-07-09T11:59:37.193067Z","iopub.status.idle":"2025-07-09T11:59:37.215746Z","shell.execute_reply.started":"2025-07-09T11:59:37.193040Z","shell.execute_reply":"2025-07-09T11:59:37.214852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:37.217312Z","iopub.execute_input":"2025-07-09T11:59:37.218484Z","iopub.status.idle":"2025-07-09T11:59:37.253938Z","shell.execute_reply.started":"2025-07-09T11:59:37.218433Z","shell.execute_reply":"2025-07-09T11:59:37.252828Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Time-Based Validation Split\n\n**Since the timestamp is there in X and y so we cannot split the data using train_test_split.\nWe need to split the training data based on time (not random split) To simulate real-world performance and avoid data leakage.**\n","metadata":{}},{"cell_type":"markdown","source":"Check if Index (Timestamp) is Sorted","metadata":{}},{"cell_type":"code","source":"# Check if X's index (timestamp) is sorted\nis_sorted = X.index.is_monotonic_increasing\nprint(\"Is X index sorted by time?\\n\", is_sorted)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:37.255270Z","iopub.execute_input":"2025-07-09T11:59:37.255778Z","iopub.status.idle":"2025-07-09T11:59:37.275974Z","shell.execute_reply.started":"2025-07-09T11:59:37.255747Z","shell.execute_reply":"2025-07-09T11:59:37.274806Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"is_y_sorted = y.index.is_monotonic_increasing\nprint(\"Is y index sorted by time?\\n\", is_y_sorted)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:37.277488Z","iopub.execute_input":"2025-07-09T11:59:37.277899Z","iopub.status.idle":"2025-07-09T11:59:37.293923Z","shell.execute_reply.started":"2025-07-09T11:59:37.277877Z","shell.execute_reply":"2025-07-09T11:59:37.292515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if not X.index.is_monotonic_increasing:\n    print(\"Sorting X and y by timestamp...\")\n    X = X.sort_index()\n    y = y.loc[X.index]\nelse:\n    print(\"Timestamps already sorted.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:37.295140Z","iopub.execute_input":"2025-07-09T11:59:37.295466Z","iopub.status.idle":"2025-07-09T11:59:37.312432Z","shell.execute_reply.started":"2025-07-09T11:59:37.295439Z","shell.execute_reply":"2025-07-09T11:59:37.310976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Use the last 10% as validation\nval_size = int(len(X) * 0.1)\n\nX_train = X.iloc[val_size:]\ny_train = y.iloc[val_size:]\nX_val = X.iloc[:val_size]\ny_val = y.iloc[:val_size]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:37.314090Z","iopub.execute_input":"2025-07-09T11:59:37.314718Z","iopub.status.idle":"2025-07-09T11:59:37.335221Z","shell.execute_reply.started":"2025-07-09T11:59:37.314673Z","shell.execute_reply":"2025-07-09T11:59:37.333823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape,y_train.shape, X_val.shape, y_val.shape, X.shape,y.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:37.340841Z","iopub.execute_input":"2025-07-09T11:59:37.341175Z","iopub.status.idle":"2025-07-09T11:59:37.357738Z","shell.execute_reply.started":"2025-07-09T11:59:37.341142Z","shell.execute_reply":"2025-07-09T11:59:37.356853Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train LightGBM Model","metadata":{}},{"cell_type":"markdown","source":"**📦 1. Import and Setup**","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom scipy.stats import pearsonr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:37.359402Z","iopub.execute_input":"2025-07-09T11:59:37.359919Z","iopub.status.idle":"2025-07-09T11:59:43.825099Z","shell.execute_reply.started":"2025-07-09T11:59:37.359878Z","shell.execute_reply":"2025-07-09T11:59:43.823812Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**🧠 2. Define Model and Fit**","metadata":{}},{"cell_type":"code","source":"lgbm_model = LGBMRegressor(\n    objective='regression',\n    learning_rate=0.02,\n    num_leaves=64,\n    n_estimators=1000,\n    feature_fraction=0.8,\n    bagging_fraction=0.8,\n    bagging_freq=5,\n    force_col_wise=True,\n    verbosity=-1,\n    random_state=42\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:43.826393Z","iopub.execute_input":"2025-07-09T11:59:43.827569Z","iopub.status.idle":"2025-07-09T11:59:43.832454Z","shell.execute_reply.started":"2025-07-09T11:59:43.827534Z","shell.execute_reply":"2025-07-09T11:59:43.831394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T11:59:43.833509Z","iopub.execute_input":"2025-07-09T11:59:43.833801Z","iopub.status.idle":"2025-07-09T12:12:33.898782Z","shell.execute_reply.started":"2025-07-09T11:59:43.833775Z","shell.execute_reply":"2025-07-09T12:12:33.897588Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**📈 3. Evaluate Pearson on Validation Set**","metadata":{}},{"cell_type":"code","source":"# Predict on validation set\nval_preds = lgbm_model.predict(X_val)\n\n# Pearson correlation (competition metric)\npearson = pearsonr(y_val, val_preds)[0]\nprint(f\"📊 Pearson Correlation on Validation: {pearson:.6f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T12:12:33.900142Z","iopub.execute_input":"2025-07-09T12:12:33.900516Z","iopub.status.idle":"2025-07-09T12:12:41.828015Z","shell.execute_reply.started":"2025-07-09T12:12:33.900486Z","shell.execute_reply":"2025-07-09T12:12:41.824901Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"df_sub = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T12:12:41.835171Z","iopub.execute_input":"2025-07-09T12:12:41.836903Z","iopub.status.idle":"2025-07-09T12:12:42.459812Z","shell.execute_reply.started":"2025-07-09T12:12:41.836788Z","shell.execute_reply":"2025-07-09T12:12:42.455434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sub.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T12:12:42.462356Z","iopub.execute_input":"2025-07-09T12:12:42.462683Z","iopub.status.idle":"2025-07-09T12:12:42.492857Z","shell.execute_reply.started":"2025-07-09T12:12:42.462655Z","shell.execute_reply":"2025-07-09T12:12:42.487630Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T12:12:42.496112Z","iopub.execute_input":"2025-07-09T12:12:42.496449Z","iopub.status.idle":"2025-07-09T12:12:42.555131Z","shell.execute_reply.started":"2025-07-09T12:12:42.496425Z","shell.execute_reply":"2025-07-09T12:12:42.554052Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.drop(columns=[\"label\"], axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T12:12:42.557843Z","iopub.execute_input":"2025-07-09T12:12:42.560246Z","iopub.status.idle":"2025-07-09T12:12:46.947868Z","shell.execute_reply.started":"2025-07-09T12:12:42.560143Z","shell.execute_reply":"2025-07-09T12:12:46.946114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on test set\ntest_preds = lgbm_model.predict(df_test)\n\n# Load sample submission and replace predictions\n\ndf_sub['prediction'] = test_preds\ndf_sub.to_csv('submission.csv', index=False)\n\nprint(\"✅ Submission saved!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T12:12:46.949528Z","iopub.execute_input":"2025-07-09T12:12:46.949938Z","iopub.status.idle":"2025-07-09T12:13:47.190366Z","shell.execute_reply.started":"2025-07-09T12:12:46.949907Z","shell.execute_reply":"2025-07-09T12:13:47.189206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sub.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T12:13:47.191550Z","iopub.execute_input":"2025-07-09T12:13:47.191897Z","iopub.status.idle":"2025-07-09T12:13:47.202865Z","shell.execute_reply.started":"2025-07-09T12:13:47.191872Z","shell.execute_reply":"2025-07-09T12:13:47.201881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sub.shape,df_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-09T12:13:47.204079Z","iopub.execute_input":"2025-07-09T12:13:47.204362Z","iopub.status.idle":"2025-07-09T12:13:47.223309Z","shell.execute_reply.started":"2025-07-09T12:13:47.204341Z","shell.execute_reply":"2025-07-09T12:13:47.222297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}