{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"},{"sourceId":204345,"sourceType":"modelInstanceVersion","modelInstanceId":174338,"modelId":196684}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:55:42.828082Z","iopub.execute_input":"2025-07-08T18:55:42.828505Z","iopub.status.idle":"2025-07-08T18:55:42.893222Z","shell.execute_reply.started":"2025-07-08T18:55:42.828478Z","shell.execute_reply":"2025-07-08T18:55:42.892235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=9/part-0.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:55:42.895172Z","iopub.execute_input":"2025-07-08T18:55:42.89567Z","iopub.status.idle":"2025-07-08T18:55:54.165682Z","shell.execute_reply.started":"2025-07-08T18:55:42.895596Z","shell.execute_reply":"2025-07-08T18:55:54.163903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#train[train.symbol_id==16],groupby(['time_id']).nunique()\ntrain['date_time'] = train.date_id*[1000] + train.time_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:55:54.168182Z","iopub.execute_input":"2025-07-08T18:55:54.168499Z","iopub.status.idle":"2025-07-08T18:55:54.20518Z","shell.execute_reply.started":"2025-07-08T18:55:54.168473Z","shell.execute_reply":"2025-07-08T18:55:54.204218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#d = train[FEAT_COLS].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:55:54.206606Z","iopub.execute_input":"2025-07-08T18:55:54.206869Z","iopub.status.idle":"2025-07-08T18:55:54.210762Z","shell.execute_reply.started":"2025-07-08T18:55:54.206848Z","shell.execute_reply":"2025-07-08T18:55:54.209747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_columns', 80)\n#d.loc[:,(d.loc['min'] > 0) [d.loc['min'] > 0].index]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:55:54.212117Z","iopub.execute_input":"2025-07-08T18:55:54.212533Z","iopub.status.idle":"2025-07-08T18:55:54.229662Z","shell.execute_reply.started":"2025-07-08T18:55:54.212501Z","shell.execute_reply":"2025-07-08T18:55:54.228095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.groupby(['symbol_id','feature_11'])['date_id'].count()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:55:54.230977Z","iopub.execute_input":"2025-07-08T18:55:54.231432Z","iopub.status.idle":"2025-07-08T18:55:54.470023Z","shell.execute_reply.started":"2025-07-08T18:55:54.231386Z","shell.execute_reply":"2025-07-08T18:55:54.468909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(10,6))\nsymbol_16 = train[train.symbol_id==10]\ne = symbol_16[symbol_16.date_id < 190]\nplt.scatter(e.date_time, e.responder_6.cumsum(), s=1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:55:54.470785Z","iopub.execute_input":"2025-07-08T18:55:54.471035Z","iopub.status.idle":"2025-07-08T18:55:54.861538Z","shell.execute_reply.started":"2025-07-08T18:55:54.471015Z","shell.execute_reply":"2025-07-08T18:55:54.859745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.loc[:,train.columns.str.startswith('feature')].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:55:54.862281Z","iopub.execute_input":"2025-07-08T18:55:54.862583Z","iopub.status.idle":"2025-07-08T18:56:15.265765Z","shell.execute_reply.started":"2025-07-08T18:55:54.862555Z","shell.execute_reply":"2025-07-08T18:56:15.264712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import polars as pl\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport joblib\n\nTARGET = 'responder_6'\nFEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]\nfrac= 0.1\n\ntest = pl.read_parquet('/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet/date_id=0/part-0.parquet')\n\ndef sample_it(s):\n    return bool(np.random.binomial(1, frac))\n\ndef load_data(date_id_range=None, time_id_range=None, columns=None, sample=1,return_type='pl'):\n\n    data_dir = '../input/jane-street-real-time-market-data-forecasting'\n    # Load data using Polars lazy loading (scan_parquet)\n    data = pl.scan_parquet(f\"{data_dir}/train.parquet\")\n\n    # Apply date_id filter if specified\n    if date_id_range is not None:\n        start_date, end_date = date_id_range\n        data = data.filter((pl.col(\"date_id\") >= start_date) & (pl.col(\"date_id\") <= end_date))\n\n    # Apply time_id filter if specified\n    if time_id_range is not None:\n        start_time, end_time = time_id_range\n        data = data.filter((pl.col(\"time_id\") >= start_time) & (pl.col(\"time_id\") <= end_time))\n\n    # Select specific columns if specified\n    if columns is not None:\n        data = data.select(columns)\n\n    # sample subset of rows using sample_it function\n    frac = sample\n    #data = data.with_columns(pl.first().map_elements(sample_it).alias(\"_sample\")).filter(pl.col(\"_sample\"))\n    data = data.gather_every(int(1/sample))\n    data = data.fill_nan(0)\n    data = data.fill_null(0)\n    # Collect the data to execute the lazy operations\n    if return_type == 'pd':\n        return data.collect().to_pandas()\n    else:\n        return data.collect()\n        \ndef calculate_r2(y_true, y_pred, weights):\n    numerator = np.sum(weights * (y_true - y_pred) ** 2)\n    denominator = np.sum(weights * (y_true ** 2))\n    r2_score = 1 - (numerator / denominator)\n    return r2_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:56:15.268496Z","iopub.execute_input":"2025-07-08T18:56:15.268832Z","iopub.status.idle":"2025-07-08T18:56:15.306553Z","shell.execute_reply.started":"2025-07-08T18:56:15.268804Z","shell.execute_reply":"2025-07-08T18:56:15.305709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import statsmodels.api as sm\noffset =  1050\ndef train_kfold_single(total_days=300, n_splits=5, save_model=True, sample = 1, save_path=\"models/\"):  \n    # Ensure save_path exists\n    if save_model and not os.path.exists(save_path):\n        os.makedirs(save_path)\n\n    # Split the time series into `n_splits` parts\n    fold_size = total_days // n_splits\n    folds = [(i * fold_size + offset, min((i + 1) * fold_size - 1, total_days - 1) + offset) for i in range(n_splits)]\n\n    cv_scores = []\n    model_group = []  # To save all models for each fold\n\n    for fold_idx in range(n_splits):\n        # The range of the current fold's validation set\n        valid_range = folds[fold_idx]\n        # The remaining parts are used as the training set\n        train_ranges = [folds[i] for i in range(n_splits) if i != fold_idx]\n\n        print(f\"Fold {fold_idx}: validation range {valid_range}, train parts: {train_ranges}\")\n\n        # Load validation data\n        valid_data = load_data(date_id_range=valid_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl', sample = sample)\n        valid_weight = valid_data['weight'].to_pandas()\n        print(\"loading\")\n        # Load training data\n        train_data = None\n        for train_range in train_ranges:\n            partial_train_data = load_data(date_id_range=train_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl',sample = sample)\n            if train_data is None:\n                train_data = partial_train_data\n            else:\n                train_data = train_data.vstack(partial_train_data)\n        train_weight = train_data['weight'].to_pandas()\n\n        # Train the model\n        #X = train_data[FEAT_COLS + ['weight']].to_pandas()\n        X = train_data[FEAT_COLS ].to_pandas()\n\n        rescale = X.std()\n        shift = X.mean()\n        X = (X - shift)/rescale\n        y_train = train_data[TARGET].to_pandas()\n        model = sm.WLS(y_train, X, weights=train_weight)\n        print(\"fitting\")\n\n        results = model.fit_regularized(alpha = 0.015, L1_wt = 0)\n        #print(results.fittedvalues)\n        r2_score_train = calculate_r2(y_train, results.fittedvalues, train_weight)\n        print(f\"Fold {fold_idx} test R2 score: {r2_score_train}\")\n        \n        #X_valid = valid_data.select(FEAT_COLS + ['weight']).to_pandas()\n        X_valid = valid_data.select(FEAT_COLS).to_pandas()\n        X_valid = (X_valid - shift)/rescale \n        y_valid_pred = results.predict(X_valid)\n        r2_score = calculate_r2(valid_data[TARGET].to_pandas(), y_valid_pred, valid_weight)\n        print(f\"Fold {fold_idx} validation R2 score: {r2_score}\")\n        \n        model_group.append(results)\n        # Save each fold result\n        cv_scores.append(r2_score)\n        break\n\n    # Model fusion: The output of all models is averaged\n    print(f\"Total trained models: {len(model_group)}\")\n    final_model = model_group[0]  # The structure of the first model is used\n    print(\"Averaging models...\")\n    average_predictions = lambda data: average_models(model_group, data)\n\n    # Save the entire model group\n    if save_model:\n        joblib.dump(final_model.params, \"linear_model.pkl\")\n        print(\"Saved the final merged model to linear_model.pkl\")\n        joblib.dump(shift, \"shift.pkl\")\n        joblib.dump(rescale, \"rescale.pkl\")\n\n\n    # Print the cross-validation results\n    print(f\"Cross-validation R2 scores: {cv_scores}\")\n    print(f\"Mean R2 score: {np.mean(cv_scores)}, Std: {np.std(cv_scores)}\")\n\n    return final_model, np.mean(cv_scores), np.std(cv_scores)\ntrain_kfold_single(total_days=600, sample = 0.14)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:56:15.308178Z","iopub.execute_input":"2025-07-08T18:56:15.308632Z","iopub.status.idle":"2025-07-08T18:57:48.941486Z","shell.execute_reply.started":"2025-07-08T18:56:15.30859Z","shell.execute_reply":"2025-07-08T18:57:48.937974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import statsmodels.api as sm\n\ndef train(train_range=(1000,1650), sample= 0.15, save_path=\"models/\"):  \n    # Ensure save_path exists\n    if not os.path.exists(save_path):\n        os.makedirs(save_path)\n\n    print(\"loading\")\n    # Load training data\n    train_data = load_data(date_id_range=train_range, columns=[\"date_id\", \"weight\"] + FEAT_COLS + [TARGET], return_type='pl',sample = sample)\n\n    train_weight = train_data['weight'].to_pandas()\n\n    # Train the model\n    #X = train_data[FEAT_COLS + ['weight']].to_pandas()\n    X = train_data[FEAT_COLS ].to_pandas()\n\n    rescale = X.std()\n    shift = X.mean()\n    X = (X - shift)/rescale\n    y_train = train_data[TARGET].to_pandas()\n    model = sm.WLS(y_train, X, weights=train_weight)\n    print(\"fitting\")\n\n    results = model.fit_regularized(alpha = 0.015, L1_wt = 0)\n    #print(results.fittedvalues)\n    r2_score_train = calculate_r2(y_train, results.fittedvalues, train_weight)\n    print(f\"Test R2 score: {r2_score_train}\")\n    \n\n    # Save the entire model group\n    joblib.dump(results.params, \"linear_model.pkl\")\n    print(\"Saved the final merged model to linear_model.pkl\")\n    joblib.dump(shift, \"shift.pkl\")\n    joblib.dump(rescale, \"rescale.pkl\")\n\n    return results\ntrain(sample = 0.15)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-08T18:57:48.944812Z","iopub.execute_input":"2025-07-08T18:57:48.945248Z","execution_failed":"2025-07-08T18:58:16.93Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import kaggle_evaluation.jane_street_inference_server\nFEAT_COLS = [f\"feature_{i:02d}\" for i in range(79)]\nimport statsmodels.api as sm\nparams = joblib.load('/kaggle/input/janestreetsimplelinear/other/default/1/linear_model.pkl')\nshift = joblib.load('/kaggle/input/janestreetsimplelinear/other/default/1/shift.pkl')\nrescale = joblib.load('/kaggle/input/janestreetsimplelinear/other/default/1/rescale.pkl')\nlags_ : pl.DataFrame | None = None\n\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n    \n    X = (test[FEAT_COLS ].fill_nan(0).fill_null(0).to_pandas() - shift) / rescale\n    # Replace this section with your own predictions\n    predictions = X.dot(params).to_frame()\n    predictions = pd.concat([test['row_id'].to_pandas().to_frame(),predictions], axis=1)\n    predictions.columns = ['row_id', 'responder_6']\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    # Confirm has as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions\n\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-08T18:58:16.931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}