{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11037875,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:30.707190Z","iopub.execute_input":"2025-02-23T20:13:30.707530Z","iopub.status.idle":"2025-02-23T20:13:30.744414Z","shell.execute_reply.started":"2025-02-23T20:13:30.707506Z","shell.execute_reply":"2025-02-23T20:13:30.743028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport polars as pl\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder, LabelEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score\nimport kaggle_evaluation.jane_street_inference_server","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:30.745807Z","iopub.execute_input":"2025-02-23T20:13:30.746208Z","iopub.status.idle":"2025-02-23T20:13:31.061342Z","shell.execute_reply.started":"2025-02-23T20:13:30.746180Z","shell.execute_reply":"2025-02-23T20:13:31.060235Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"features_data = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv')\nresponders_data = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv')\nsample_data = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/sample_submission.csv')\ntrain_data = pl.scan_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\").filter(pl.col(\"partition_id\") == 0).collect()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:31.062991Z","iopub.execute_input":"2025-02-23T20:13:31.063762Z","iopub.status.idle":"2025-02-23T20:13:31.780737Z","shell.execute_reply.started":"2025-02-23T20:13:31.063732Z","shell.execute_reply":"2025-02-23T20:13:31.779466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(features_data.dtypes)\nprint(responders_data.dtypes)\nprint(sample_data.dtypes)\n#check data types","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:31.782475Z","iopub.execute_input":"2025-02-23T20:13:31.782859Z","iopub.status.idle":"2025-02-23T20:13:31.792407Z","shell.execute_reply.started":"2025-02-23T20:13:31.782831Z","shell.execute_reply":"2025-02-23T20:13:31.791034Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# encoding\nlabelencoder = LabelEncoder()\nfeatures_data['feature'] = labelencoder.fit_transform(features_data['feature'])\nresponders_data['responder'] = labelencoder.fit_transform(responders_data['responder'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:31.793677Z","iopub.execute_input":"2025-02-23T20:13:31.794021Z","iopub.status.idle":"2025-02-23T20:13:31.814098Z","shell.execute_reply.started":"2025-02-23T20:13:31.793993Z","shell.execute_reply":"2025-02-23T20:13:31.812697Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis(EDA)","metadata":{}},{"cell_type":"code","source":"def perform_eda(df, name):\n    print(f\"\\n{name} Dataset Shape: {df.shape}\")\n    print(\"\\nMissing Values:\\n\", df.isnull().sum().sort_values(ascending=False))\n    print(\"\\nSummary Statistics:\\n\", df.describe())\n    \n    # correlation heatmap\n    plt.figure(figsize=(10, 8))\n    sns.heatmap(df.corr(), annot=True, cmap='coolwarm', fmt=\".2f\")\n    plt.title(f'Correlation Heatmap of {name}')\n    plt.show()\n    \n    # pie chart for categorical variables\n    cat_cols = df.select_dtypes(include=['object', 'bool']).columns\n    for col in cat_cols:\n        plt.figure(figsize=(6, 6))\n        df[col].value_counts().plot.pie(autopct='%1.1f%%')\n        plt.title(f'Distribution of {col}')\n        plt.ylabel('')\n        plt.show()\n\nperform_eda(features_data, \"Features\")\nperform_eda(responders_data, \"Responders\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:31.815162Z","iopub.execute_input":"2025-02-23T20:13:31.815457Z","iopub.status.idle":"2025-02-23T20:13:35.778064Z","shell.execute_reply.started":"2025-02-23T20:13:31.815431Z","shell.execute_reply":"2025-02-23T20:13:35.776868Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Converting Categorical Features Using Label Encoding","metadata":{}},{"cell_type":"code","source":"labelencoder = LabelEncoder()\nfor col in features_data.columns:\n    features_data[col] = labelencoder.fit_transform(features_data[col])\nfor col in responders_data.columns:\n    responders_data[col] = labelencoder.fit_transform(responders_data[col])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:35.779341Z","iopub.execute_input":"2025-02-23T20:13:35.779781Z","iopub.status.idle":"2025-02-23T20:13:35.802116Z","shell.execute_reply.started":"2025-02-23T20:13:35.779741Z","shell.execute_reply":"2025-02-23T20:13:35.800821Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Defining Target Variable and Features","metadata":{}},{"cell_type":"code","source":"X = train_data.drop(['partition_id', 'responder_0', 'responder_1', 'responder_2', 'responder_3', 'responder_4',\n                     'responder_5', 'responder_6', 'responder_7', 'responder_8'])\ny = train_data['responder_6']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:35.805775Z","iopub.execute_input":"2025-02-23T20:13:35.806199Z","iopub.status.idle":"2025-02-23T20:13:35.830865Z","shell.execute_reply.started":"2025-02-23T20:13:35.806167Z","shell.execute_reply":"2025-02-23T20:13:35.829130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_sample = X.sample(fraction=0.1, seed=42)\nsample_indices = X_sample.get_column('date_id').to_numpy().astype(int)\ny_sample=y[sample_indices]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:35.833461Z","iopub.execute_input":"2025-02-23T20:13:35.834013Z","iopub.status.idle":"2025-02-23T20:13:36.122960Z","shell.execute_reply.started":"2025-02-23T20:13:35.833944Z","shell.execute_reply":"2025-02-23T20:13:36.121792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_sample_pd = X_sample.to_pandas()\ny_sample_pd = y_sample.to_pandas()\ny_sample_pd.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:36.124057Z","iopub.execute_input":"2025-02-23T20:13:36.124359Z","iopub.status.idle":"2025-02-23T20:13:36.224732Z","shell.execute_reply.started":"2025-02-23T20:13:36.124334Z","shell.execute_reply":"2025-02-23T20:13:36.223607Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessors","metadata":{"execution":{"iopub.status.busy":"2025-02-23T16:50:33.055072Z","iopub.execute_input":"2025-02-23T16:50:33.057355Z","iopub.status.idle":"2025-02-23T16:50:33.076643Z","shell.execute_reply.started":"2025-02-23T16:50:33.057087Z","shell.execute_reply":"2025-02-23T16:50:33.071572Z"}}},{"cell_type":"code","source":"numeric_features = X_sample_pd.select_dtypes(include=[np.number]).columns.tolist()\ncategorical_features = X_sample_pd.select_dtypes(exclude=[np.number]).columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:36.225978Z","iopub.execute_input":"2025-02-23T20:13:36.226391Z","iopub.status.idle":"2025-02-23T20:13:36.242364Z","shell.execute_reply.started":"2025-02-23T20:13:36.226350Z","shell.execute_reply":"2025-02-23T20:13:36.241266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy = 'mean')),\n    ('scaler', StandardScaler())\n]) \n\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy = 'most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown = 'ignore'))\n])\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        (\"num\", numeric_transformer, numeric_features),\n        ('cat', categorical_transformer, categorical_features)\n    ]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:36.243492Z","iopub.execute_input":"2025-02-23T20:13:36.243914Z","iopub.status.idle":"2025-02-23T20:13:36.251539Z","shell.execute_reply.started":"2025-02-23T20:13:36.243867Z","shell.execute_reply":"2025-02-23T20:13:36.250573Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Pipeline","metadata":{}},{"cell_type":"code","source":"params={'learning_rate': 0.05,\n        'objective':'regression',\n        'metric':'rmse',\n        'num_leaves': 200,\n        'verbose': 1,\n        \"subsample\": 0.99,\n        \"colsample_bytree\": 0.99,\n        \"random_state\":33,\n        'max_depth': 14,\n        'lambda_l2': 0.02085548700474218,\n        'lambda_l1': 0.004107624022751344,\n        'bagging_fraction': 0.7934712636944741,\n        'feature_fraction': 0.686612409641711,\n        'min_child_samples': 21\n       }\npipeline = Pipeline(steps=[\n    (\"preprocessor\", preprocessor),\n    (\"model\", LGBMRegressor(**params))\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:36.252886Z","iopub.execute_input":"2025-02-23T20:13:36.253340Z","iopub.status.idle":"2025-02-23T20:13:36.278863Z","shell.execute_reply.started":"2025-02-23T20:13:36.253297Z","shell.execute_reply":"2025-02-23T20:13:36.277539Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train-Test Split","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X_sample_pd, y_sample_pd, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:36.280148Z","iopub.execute_input":"2025-02-23T20:13:36.280535Z","iopub.status.idle":"2025-02-23T20:13:36.399841Z","shell.execute_reply.started":"2025-02-23T20:13:36.280498Z","shell.execute_reply":"2025-02-23T20:13:36.398666Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"X_train_transformed = pipeline.named_steps['preprocessor'].fit_transform(X_train)\nX_test_transformed = pipeline.named_steps['preprocessor'].transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:36.401095Z","iopub.execute_input":"2025-02-23T20:13:36.401449Z","iopub.status.idle":"2025-02-23T20:13:37.211428Z","shell.execute_reply.started":"2025-02-23T20:13:36.401406Z","shell.execute_reply":"2025-02-23T20:13:37.210432Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train Model ","metadata":{}},{"cell_type":"code","source":"model = LGBMRegressor(**params, n_estimators=1000)\nmodel.fit(\n    X_train_transformed, y_train, \n    eval_set=[(X_test_transformed, y_test)], \n    eval_metric='rmse', \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:13:37.212419Z","iopub.execute_input":"2025-02-23T20:13:37.212692Z","iopub.status.idle":"2025-02-23T20:14:41.852704Z","shell.execute_reply.started":"2025-02-23T20:13:37.212669Z","shell.execute_reply":"2025-02-23T20:14:41.851485Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipeline.named_steps['model'] = model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:14:41.853755Z","iopub.execute_input":"2025-02-23T20:14:41.854171Z","iopub.status.idle":"2025-02-23T20:14:41.859396Z","shell.execute_reply.started":"2025-02-23T20:14:41.854133Z","shell.execute_reply":"2025-02-23T20:14:41.858032Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Evaluation","metadata":{}},{"cell_type":"code","source":"pipeline.named_steps['model'].fit(X_train_transformed, y_train)\npredictions = pipeline.named_steps['model'].predict(X_test_transformed)\nmse = mean_squared_error(y_test, predictions)\nrmse = np.sqrt(mse)\nr2 = r2_score(y_test, predictions)\n\nprint(f\"MSE: {mse}\\nRMSE: {rmse}\\nR2 Score: {r2}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:14:41.860600Z","iopub.execute_input":"2025-02-23T20:14:41.860975Z","iopub.status.idle":"2025-02-23T20:14:52.060695Z","shell.execute_reply.started":"2025-02-23T20:14:41.860938Z","shell.execute_reply":"2025-02-23T20:14:52.059587Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot Predictions\nplt.scatter(y_test, predictions)\nplt.xlabel('Actual Values')\nplt.ylabel('Predicted Values')\nplt.title('Predictions vs Actuals')\nz = np.polyfit(y_test, predictions, 1)\np = np.poly1d(z)\nplt.plot(y_test, p(y_test), color='magenta')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-23T20:14:52.062098Z","iopub.execute_input":"2025-02-23T20:14:52.062525Z","iopub.status.idle":"2025-02-23T20:14:52.371752Z","shell.execute_reply.started":"2025-02-23T20:14:52.062483Z","shell.execute_reply":"2025-02-23T20:14:52.370411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}