{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:45:34.097006Z","iopub.execute_input":"2025-12-25T09:45:34.097811Z","iopub.status.idle":"2025-12-25T09:45:34.147147Z","shell.execute_reply.started":"2025-12-25T09:45:34.097774Z","shell.execute_reply":"2025-12-25T09:45:34.145709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_columns',None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:45:34.149223Z","iopub.execute_input":"2025-12-25T09:45:34.149554Z","iopub.status.idle":"2025-12-25T09:45:34.155214Z","shell.execute_reply.started":"2025-12-25T09:45:34.149523Z","shell.execute_reply":"2025-12-25T09:45:34.153870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_path = \"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet\"\ndf_sample = pd.read_parquet(sample_path)\n\nprint(\"Shape:\", df_sample.shape)\ndf_sample.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:45:34.156295Z","iopub.execute_input":"2025-12-25T09:45:34.156603Z","iopub.status.idle":"2025-12-25T09:45:36.164774Z","shell.execute_reply.started":"2025-12-25T09:45:34.156576Z","shell.execute_reply":"2025-12-25T09:45:36.163781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df_sample.filter(regex=\"^feature_\")\ny = df_sample[\"responder_6\"]\nw = df_sample[\"weight\"]\n\nprint(\"X shape:\", X.shape)\nprint(\"y shape:\", y.shape)\nprint(\"weight shape:\", w.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:45:36.165852Z","iopub.execute_input":"2025-12-25T09:45:36.166164Z","iopub.status.idle":"2025-12-25T09:45:36.421936Z","shell.execute_reply.started":"2025-12-25T09:45:36.166131Z","shell.execute_reply":"2025-12-25T09:45:36.420982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# considering only only 200,000 rows as teh df_sample is very large (1.9 million rows)\n\ndf_200krows = df_sample.sample(n=200_000, random_state=42)\n\nprint(\"Original rows:\", df_sample.shape[0])\nprint(\"Sampled rows:\", df_200krows.shape[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:45:36.424140Z","iopub.execute_input":"2025-12-25T09:45:36.424591Z","iopub.status.idle":"2025-12-25T09:45:36.652658Z","shell.execute_reply.started":"2025-12-25T09:45:36.424563Z","shell.execute_reply":"2025-12-25T09:45:36.651774Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train-test split (simple, random)\n\nfrom sklearn.model_selection import train_test_split\n\nX = df_200krows.filter(regex=\"^feature_\")\ny = df_200krows[\"responder_6\"]\nw = df_200krows[\"weight\"]\n\nX_train, X_test, y_train, y_test, w_train, w_test = train_test_split(X, y, w, test_size=0.2, random_state=42)\n\nprint(\"X_train shape:\", X_train.shape)\nprint(\"X_test shape:\", X_test.shape)\nprint(\"y_train shape:\", y_train.shape)\nprint(\"y_test shape:\", y_test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:45:36.653536Z","iopub.execute_input":"2025-12-25T09:45:36.653798Z","iopub.status.idle":"2025-12-25T09:45:36.763677Z","shell.execute_reply.started":"2025-12-25T09:45:36.653774Z","shell.execute_reply":"2025-12-25T09:45:36.762581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training using random forest model\n\nfrom sklearn.ensemble import RandomForestRegressor\nmodel = RandomForestRegressor(n_estimators=50, max_depth=10, random_state=42, n_jobs=-1)\n\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:45:36.764777Z","iopub.execute_input":"2025-12-25T09:45:36.765127Z","iopub.status.idle":"2025-12-25T09:52:31.333865Z","shell.execute_reply.started":"2025-12-25T09:45:36.765100Z","shell.execute_reply":"2025-12-25T09:52:31.332933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(X_test)\n\nprint(\"First 10 predictions:\")\nprint(y_pred[:10])\n\nprint(\"\\nFirst 10 actual values:\")\nprint(y_test.iloc[:10].values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:52:31.335058Z","iopub.execute_input":"2025-12-25T09:52:31.335411Z","iopub.status.idle":"2025-12-25T09:52:31.426316Z","shell.execute_reply.started":"2025-12-25T09:52:31.335385Z","shell.execute_reply":"2025-12-25T09:52:31.425243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now I'll use a fast LightGBM model as it handles NaNs automatically and is much faster than RandomForest and also safe for real-time inference\n\nimport lightgbm as lgb\n\nX = df_200krows.filter(regex=\"^feature_\")\ny = df_200krows[\"responder_6\"]\nw = df_200krows[\"weight\"]\n\ntrain_data = lgb.Dataset(X, label=y,weight=w)\nparams = {\"objective\": \"regression\", \"learning_rate\": 0.05, \"num_leaves\": 64, \"verbosity\": -1}\n\nlgb_model = lgb.train(params, train_data, num_boost_round=100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:52:31.427407Z","iopub.execute_input":"2025-12-25T09:52:31.427690Z","iopub.status.idle":"2025-12-25T09:52:38.106417Z","shell.execute_reply.started":"2025-12-25T09:52:31.427656Z","shell.execute_reply":"2025-12-25T09:52:38.105736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# predict function \n\ndef predict(test, lags):\n    X_test = test[[col for col in test.columns if col.startswith(\"feature_\")]]\n    preds = lgb_model.predict(X_test)\n\n    return pd.DataFrame({\n        \"row_id\": test[\"row_id\"],\n        \"responder_6\": preds\n    })\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:52:38.107220Z","iopub.execute_input":"2025-12-25T09:52:38.107475Z","iopub.status.idle":"2025-12-25T09:52:38.113566Z","shell.execute_reply.started":"2025-12-25T09:52:38.107449Z","shell.execute_reply":"2025-12-25T09:52:38.112703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import kaggle_evaluation.jane_street_inference_server as js\n\njs.JSInferenceServer(predict).serve()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-25T09:52:38.114905Z","iopub.execute_input":"2025-12-25T09:52:38.115229Z","iopub.status.idle":"2025-12-25T09:52:38.137503Z","shell.execute_reply.started":"2025-12-25T09:52:38.115202Z","shell.execute_reply":"2025-12-25T09:52:38.136163Z"}},"outputs":[],"execution_count":null}]}