{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport random\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom contextlib import contextmanager\nfrom time import time\nfrom tqdm import tqdm\nimport lightgbm as lgb\nimport category_encoders as ce\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.metrics import classification_report, log_loss, accuracy_score\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import KFold, train_test_split\nimport optuna\nimport yaml\nimport polars as pl\nfrom sklearn.ensemble import RandomForestRegressor\nimport kaggle_evaluation.jane_street_inference_server\n\nfrom sklearn.ensemble import HistGradientBoostingClassifier\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.model_selection import train_test_split\n\nimport polars as pl\nimport numpy as np\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-13T17:22:44.003771Z","iopub.execute_input":"2024-11-13T17:22:44.004254Z","iopub.status.idle":"2024-11-13T17:23:05.126458Z","shell.execute_reply.started":"2024-11-13T17:22:44.004206Z","shell.execute_reply":"2024-11-13T17:23:05.125025Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_0 = pd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet')\ntest = pd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet')\n\ntrain_0 = train_0.sample(frac=1, random_state=42).reset_index(drop=True)\n\ntest_cols_0=test.columns.tolist()\ntest_cols=[ 'date_id', 'time_id', 'symbol_id', 'weight',  'feature_00', 'feature_01', 'feature_02', 'feature_03', 'feature_04', 'feature_05', 'feature_06', 'feature_07', 'feature_08', 'feature_09', 'feature_10', 'feature_11', 'feature_12', 'feature_13', 'feature_14', 'feature_15', 'feature_16', 'feature_17', 'feature_18', 'feature_19', 'feature_20', 'feature_21', 'feature_22', 'feature_23', 'feature_24', 'feature_25', 'feature_26', 'feature_27', 'feature_28', 'feature_29', 'feature_30', 'feature_31', 'feature_32', 'feature_33', 'feature_34', 'feature_35', 'feature_36', 'feature_37', 'feature_38', 'feature_39', 'feature_40', 'feature_41', 'feature_42', 'feature_43', 'feature_44', 'feature_45', 'feature_46', 'feature_47', 'feature_48', 'feature_49', 'feature_50', 'feature_51', 'feature_52', 'feature_53', 'feature_54', 'feature_55', 'feature_56', 'feature_57', 'feature_58', 'feature_59', 'feature_60', 'feature_61', 'feature_62', 'feature_63', 'feature_64', 'feature_65', 'feature_66', 'feature_67', 'feature_68', 'feature_69', 'feature_70', 'feature_71', 'feature_72', 'feature_73', 'feature_74', 'feature_75', 'feature_76', 'feature_77', 'feature_78']\ntarget='responder_6'\ndataX=train_0[test_cols]\ndataY=train_0[target]\nTEST_X=test[test_cols]","metadata":{"execution":{"iopub.status.busy":"2024-11-13T17:23:05.129014Z","iopub.execute_input":"2024-11-13T17:23:05.13018Z","iopub.status.idle":"2024-11-13T17:23:12.399215Z","shell.execute_reply.started":"2024-11-13T17:23:05.130099Z","shell.execute_reply":"2024-11-13T17:23:12.396837Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_columns = list(dataX.columns)\nprint(df_columns)\n\nm=len(dataX)\nprint(m)\nM=list(range(m))\nrandom.seed(2021)\nrandom.shuffle(M)\n\ntrainX=dataX.iloc[M[0:(m//5)*4]]\ntrainY=dataY.iloc[M[0:(m//5)*4]]\ntestX=dataX.iloc[M[(m//5)*4:]]\ntestY=dataY.iloc[M[(m//5)*4:]]\n\ntrain_df=trainX\ntest_df=testX\n\ntrain_df.columns=df_columns\ntest_df.columns=df_columns\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T17:23:12.401408Z","iopub.execute_input":"2024-11-13T17:23:12.402107Z","iopub.status.idle":"2024-11-13T17:23:17.368229Z","shell.execute_reply.started":"2024-11-13T17:23:12.402041Z","shell.execute_reply":"2024-11-13T17:23:17.366489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_from = 1455 # for private you should change to 1455 (1 year)\nall_train_data = pl.scan_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\")\nvalid_df = all_train_data.filter(pl.col(\"date_id\")>=valid_from).collect()\nvalid_df = valid_df.with_columns(pl.Series(range(len(valid_df))).alias(\"row_id\"))\nweights = valid_df.select(\"weight\").to_numpy().reshape(-1)","metadata":{"execution":{"iopub.status.busy":"2024-11-13T17:23:17.371404Z","iopub.execute_input":"2024-11-13T17:23:17.371874Z","iopub.status.idle":"2024-11-13T17:23:27.526644Z","shell.execute_reply.started":"2024-11-13T17:23:17.37183Z","shell.execute_reply":"2024-11-13T17:23:27.525453Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract target and features\ntarget_column = \"responder_6\"  # Change as appropriate\nfeatures = [col for col in valid_df.columns if col not in [target_column, \"date_id\", \"weight\"]]\n\n# Prepare X and y\nX = valid_df.select(features).to_numpy()\ny = valid_df[target_column].to_numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T17:23:27.528133Z","iopub.execute_input":"2024-11-13T17:23:27.528491Z","iopub.status.idle":"2024-11-13T17:23:33.867614Z","shell.execute_reply.started":"2024-11-13T17:23:27.528454Z","shell.execute_reply":"2024-11-13T17:23:33.86623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Extract target and features\ntarget_column = \"responder_6\"  # Change as appropriate\nfeatures = [col for col in valid_df.columns if col not in [target_column, \"date_id\", \"weight\"]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T17:23:33.870071Z","iopub.execute_input":"2024-11-13T17:23:33.87059Z","iopub.status.idle":"2024-11-13T17:23:33.886276Z","shell.execute_reply.started":"2024-11-13T17:23:33.870534Z","shell.execute_reply":"2024-11-13T17:23:33.884771Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Target (y) unique values:\", np.unique(y))\nprint(\"Target (y) data type:\", y.dtype)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T17:23:33.887762Z","iopub.execute_input":"2024-11-13T17:23:33.888121Z","iopub.status.idle":"2024-11-13T17:23:35.115879Z","shell.execute_reply.started":"2024-11-13T17:23:33.888084Z","shell.execute_reply":"2024-11-13T17:23:35.114595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = y.astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T17:23:35.117345Z","iopub.execute_input":"2024-11-13T17:23:35.117727Z","iopub.status.idle":"2024-11-13T17:23:35.154732Z","shell.execute_reply.started":"2024-11-13T17:23:35.117671Z","shell.execute_reply":"2024-11-13T17:23:35.152097Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = np.where(y > np.median(y), 1, 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T17:23:35.158129Z","iopub.execute_input":"2024-11-13T17:23:35.158992Z","iopub.status.idle":"2024-11-13T17:23:35.319628Z","shell.execute_reply.started":"2024-11-13T17:23:35.158927Z","shell.execute_reply":"2024-11-13T17:23:35.318284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Ensure X and y are still in correct form\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T17:23:35.323288Z","iopub.execute_input":"2024-11-13T17:23:35.323729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = np.where(y > np.median(y), 1, 0)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.ensemble import RandomForestClassifier, HistGradientBoostingClassifier\nfrom sklearn.model_selection import train_test_split\nimport optuna\nimport yaml\nimport joblib\nfrom sklearn.metrics import accuracy_score\nimport numpy as np\n\n# Binning configuration\nbinary_classification = True  # Set to False if you want multi-class classification\n\n# Discretize the target variable `trainY`\nif binary_classification:\n    # Binary classification using median\n    target = np.where(trainY > np.median(trainY), 1, 0)\nelse:\n    # Multi-class classification using custom bins\n    # Define bin edges, adjust as needed for your data distribution\n    bins = np.percentile(trainY, [0, 25, 50, 75, 100])\n    target = np.digitize(trainY, bins=bins) - 1  # Shift bins for zero-based classes\n\n# Define the objective function for Optuna\ndef objective_rf(trial, data=trainX, target=target):\n    train_x, test_x, train_y, test_y = train_test_split(data, target, test_size=0.2, random_state=42)\n\n    # Define the search space for RandomForestClassifier hyperparameters\n    param = {\n        'n_estimators': trial.suggest_int('n_estimators', 50, 200),\n        'max_depth': trial.suggest_int('max_depth', 5, 15),\n        'min_samples_split': trial.suggest_int('min_samples_split', 2, 10),\n        'min_samples_leaf': trial.suggest_int('min_samples_leaf', 1, 5),\n        'max_features': trial.suggest_categorical('max_features', ['auto', 'sqrt', 'log2']),\n        'bootstrap': trial.suggest_categorical('bootstrap', [True, False]),\n        'random_state': 42,\n    }\n\n    # Create a pipeline that includes imputation for NaN handling\n    model = Pipeline([\n        ('imputer', SimpleImputer(strategy='mean')),  # Replace NaNs with mean value\n        ('classifier', RandomForestClassifier(**param))\n    ])\n    model.fit(train_x, train_y)\n    \n    # Make predictions and calculate the evaluation metric\n    preds = model.predict(test_x)\n    accuracy = accuracy_score(test_y, preds)\n    return -accuracy  # Optuna minimizes, so we negate accuracy\n\n# Choose classifier\nclassifier = \"RandomForest\"  # Set to \"RandomForest\" or \"HistGradientBoosting\"\n\n# Optimize with Optuna\nfor i in range(1):\n    if classifier == \"RandomForest\":\n        objective = objective_rf\n    else:\n        # Objective for HistGradientBoostingClassifier\n        def objective_hgb(trial, data=trainX, target=target):\n            train_x, test_x, train_y, test_y = train_test_split(data, target, test_size=0.2, random_state=42)\n            param = {\n                'max_iter': trial.suggest_int('max_iter', 100, 300),\n                'max_depth': trial.suggest_int('max_depth', 5, 15),\n                'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n                'l2_regularization': trial.suggest_loguniform('l2_regularization', 0.0001, 1.0),\n                'max_bins': trial.suggest_int('max_bins', 128, 255),\n                'random_state': 42,\n            }\n            model = HistGradientBoostingClassifier(**param)\n            model.fit(train_x, train_y)\n            preds = model.predict(test_x)\n            accuracy = accuracy_score(test_y, preds)\n            return -accuracy  # Negate for minimization\n\n        objective = objective_hgb\n    \n    study = optuna.create_study(direction='minimize')\n    study.optimize(objective, n_trials=20, timeout=12*60*60)\n    \n    # Save the best model parameters to YAML\n    best_trial_params = study.best_trial.params\n    with open(f\"Best_trial_{classifier}_{i}.yaml\", \"w\") as yaml_file:\n        yaml.dump(best_trial_params, yaml_file)\n\n    # Train and save the final model with best parameters\n    if classifier == \"RandomForest\":\n        # Use pipeline with imputation for RandomForest\n        best_model = Pipeline([\n            ('imputer', SimpleImputer(strategy='constant')),  # Impute missing values\n            ('classifier', RandomForestClassifier(**best_trial_params, random_state=42))\n        ])\n    else:\n        best_model = HistGradientBoostingClassifier(**best_trial_params, random_state=42)\n\n    # Fit the model on the entire dataset\n    best_model.fit(data, target)\n\n    # Save the trained model for later serving\n    model_filename = f\"best_{classifier}_model_{i}.joblib\"\n    joblib.dump(best_model, model_filename)\n    print(f\"Model saved as {model_filename}\")\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}