{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import r2_score\nfrom sklearn.preprocessing import MinMaxScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:51:38.491357Z","iopub.execute_input":"2024-12-10T13:51:38.491859Z","iopub.status.idle":"2024-12-10T13:51:38.498896Z","shell.execute_reply.started":"2024-12-10T13:51:38.491815Z","shell.execute_reply":"2024-12-10T13:51:38.497408Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"markdown","source":"Loading a randomly selected input file, e.g. `train.parquet/partition_id=1/part-0.parquet`.","metadata":{}},{"cell_type":"code","source":"input_path = '/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet'\nfilename = 'partition_id=5/part-0.parquet'\ndf = pd.read_parquet(os.path.join(input_path, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:51:42.601150Z","iopub.execute_input":"2024-12-10T13:51:42.601584Z","iopub.status.idle":"2024-12-10T13:51:54.639512Z","shell.execute_reply.started":"2024-12-10T13:51:42.601547Z","shell.execute_reply":"2024-12-10T13:51:54.637940Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:51:59.304572Z","iopub.execute_input":"2024-12-10T13:51:59.305358Z","iopub.status.idle":"2024-12-10T13:51:59.315151Z","shell.execute_reply.started":"2024-12-10T13:51:59.305308Z","shell.execute_reply":"2024-12-10T13:51:59.313868Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:52:00.280228Z","iopub.execute_input":"2024-12-10T13:52:00.280765Z","iopub.status.idle":"2024-12-10T13:52:00.319117Z","shell.execute_reply.started":"2024-12-10T13:52:00.280717Z","shell.execute_reply":"2024-12-10T13:52:00.317452Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Linear regression","metadata":{}},{"cell_type":"markdown","source":"Training a linear regression model to predict values in `responder_6`, using values in `feature_xx` (because why not).","metadata":{}},{"cell_type":"code","source":"# Filtering on 'features_xx'\nX = df[[col for col in df.columns if col.startswith('feature_')]]\n\n# replacing NaN with 0\nX = X.fillna(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:52:04.372930Z","iopub.execute_input":"2024-12-10T13:52:04.373599Z","iopub.status.idle":"2024-12-10T13:52:07.299943Z","shell.execute_reply.started":"2024-12-10T13:52:04.373541Z","shell.execute_reply":"2024-12-10T13:52:07.298845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = df[['responder_6']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:52:07.301956Z","iopub.execute_input":"2024-12-10T13:52:07.302463Z","iopub.status.idle":"2024-12-10T13:52:07.314478Z","shell.execute_reply.started":"2024-12-10T13:52:07.302412Z","shell.execute_reply":"2024-12-10T13:52:07.313002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train-test splits","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.25)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:52:09.574832Z","iopub.execute_input":"2024-12-10T13:52:09.575256Z","iopub.status.idle":"2024-12-10T13:52:15.625140Z","shell.execute_reply.started":"2024-12-10T13:52:09.575221Z","shell.execute_reply":"2024-12-10T13:52:15.622916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = MinMaxScaler()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:52:15.628305Z","iopub.execute_input":"2024-12-10T13:52:15.628902Z","iopub.status.idle":"2024-12-10T13:52:15.635187Z","shell.execute_reply.started":"2024-12-10T13:52:15.628848Z","shell.execute_reply":"2024-12-10T13:52:15.633837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_norm = scaler.fit_transform(X_train)\nX_test_norm = scaler.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:52:15.637027Z","iopub.execute_input":"2024-12-10T13:52:15.637644Z","iopub.status.idle":"2024-12-10T13:52:18.320213Z","shell.execute_reply.started":"2024-12-10T13:52:15.637362Z","shell.execute_reply":"2024-12-10T13:52:18.318935Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Training model","metadata":{}},{"cell_type":"code","source":"model = LinearRegression()\nmodel.fit(X_train_norm, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:52:20.041472Z","iopub.execute_input":"2024-12-10T13:52:20.041954Z","iopub.status.idle":"2024-12-10T13:52:45.420658Z","shell.execute_reply.started":"2024-12-10T13:52:20.041915Z","shell.execute_reply":"2024-12-10T13:52:45.419294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred_train = model.predict(X_train_norm)\ny_pred_test = model.predict(X_test_norm)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:52:45.422592Z","iopub.execute_input":"2024-12-10T13:52:45.422956Z","iopub.status.idle":"2024-12-10T13:52:45.725689Z","shell.execute_reply.started":"2024-12-10T13:52:45.422921Z","shell.execute_reply":"2024-12-10T13:52:45.723429Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Evaluation","metadata":{}},{"cell_type":"code","source":"print(f'Training r2: {r2_score(y_train, y_pred_train)}')\nprint(f'Test r2: {r2_score(y_test, y_pred_test)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:52:47.123354Z","iopub.execute_input":"2024-12-10T13:52:47.123759Z","iopub.status.idle":"2024-12-10T13:52:47.168337Z","shell.execute_reply.started":"2024-12-10T13:52:47.123726Z","shell.execute_reply":"2024-12-10T13:52:47.167011Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submit to server","metadata":{}},{"cell_type":"code","source":"import polars as pl\n\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:52:50.586333Z","iopub.execute_input":"2024-12-10T13:52:50.586748Z","iopub.status.idle":"2024-12-10T13:52:50.592571Z","shell.execute_reply.started":"2024-12-10T13:52:50.586715Z","shell.execute_reply":"2024-12-10T13:52:50.591266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n    \n    # Replace this section with your own predictions\n    predictions = test.select(\n        'row_id',\n        pl.lit(0.0).alias('responder_6'),\n    )\n    #mypreds = test['weight'] + 1\n    #predictions = predictions.with_columns(pl.Series('responder_6', mypreds))\n\n    X = test[[col for col in test.columns if col.startswith('feature_')]]\n    X = X.fill_null(0)\n    X_norm = scaler.transform(X)\n    y_pred = model.predict(X_norm)\n    predictions = predictions.with_columns(pl.Series('responder_6', y_pred.ravel()))\n\n    if isinstance(predictions, pl.DataFrame):\n        assert predictions.columns == ['row_id', 'responder_6']\n    elif isinstance(predictions, pd.DataFrame):\n        assert (predictions.columns == ['row_id', 'responder_6']).all()\n    else:\n        raise TypeError('The predict function must return a DataFrame')\n    # Confirm has as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:53:00.224085Z","iopub.execute_input":"2024-12-10T13:53:00.224480Z","iopub.status.idle":"2024-12-10T13:53:00.234986Z","shell.execute_reply.started":"2024-12-10T13:53:00.224448Z","shell.execute_reply":"2024-12-10T13:53:00.233348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"inference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T13:53:01.585995Z","iopub.execute_input":"2024-12-10T13:53:01.586384Z","iopub.status.idle":"2024-12-10T13:53:01.642406Z","shell.execute_reply.started":"2024-12-10T13:53:01.586352Z","shell.execute_reply":"2024-12-10T13:53:01.641329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}