{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":11305158,"sourceType":"competition"},{"sourceId":203900450,"sourceType":"kernelVersion"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Libs Needed","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n###################################\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n###################################\nfrom sklearn.model_selection import train_test_split, TimeSeriesSplit\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import r2_score, mean_absolute_error, mean_squared_error\n###################################\nfrom sklearn.preprocessing import MinMaxScaler\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout\nfrom sklearn.model_selection import train_test_split\n###################\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n#################\nimport polars as pl","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T23:55:33.084815Z","iopub.execute_input":"2025-05-27T23:55:33.085587Z","iopub.status.idle":"2025-05-27T23:55:46.906183Z","shell.execute_reply.started":"2025-05-27T23:55:33.085558Z","shell.execute_reply":"2025-05-27T23:55:46.905081Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Configuration, Data Load, & Data Split\n  - Note: The LSTM model was trained but not blended in the final ensemble\n    - It tends to overfit for JS24 data, some kagglers guessed that\n    - some features (possible signals) includes the lag info or created by time series\n    - methods \n  - LSTM can be regarded as the most basic NN approach for time series analysis\n  - It resembles the Hidden Markov models in traditional stats","metadata":{}},{"cell_type":"code","source":"# Configuration\n\nclass CFG: \n    # Note: this is convenient for \n    # updating data/versioning due to different input\n    # which is a very common use in Kaggle community\n    seed = 42\n    target_col = \"responder_6\"\n    feature_cols = [\"symbol_id\", \"time_id\"] \\\n        + [f\"feature_{idx:02d}\" for idx in range(79)]\n    ##########\n    categorical_cols = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T23:55:46.908564Z","iopub.execute_input":"2025-05-27T23:55:46.909247Z","iopub.status.idle":"2025-05-27T23:55:46.915262Z","shell.execute_reply.started":"2025-05-27T23:55:46.909219Z","shell.execute_reply":"2025-05-27T23:55:46.914119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data loading\n\nDT_GT = 1550 \n# Note: Kaggle doesn't have enough RAM for full-size data\n# Full-size can be done using chunk-wise run in Colab Pro+\n\ntrain = pl.scan_parquet(\n    \"/kaggle/input/js24-preprocessing-create-lags/training.parquet\"\n).filter(pl.col(\"date_id\") > DT_GT).collect().to_pandas()\n\nvalid = pl.scan_parquet(\n    \"/kaggle/input/js24-preprocessing-create-lags/validation.parquet\"\n).filter(pl.col(\"date_id\") > DT_GT).collect().to_pandas()\n\n# train.shape, valid.shape\ntrain = pd.concat([train, valid]).reset_index(drop=True)\ntrain = train.fillna(method = 'ffill').fillna(0)\nvalid = valid.fillna(method = 'ffill').fillna(0)\n\n# Train vs Valid (We do One-fold in this demo)\n\nX_train = train[ CFG.feature_cols ]\ny_train = train[ CFG.target_col ]\nw_train = train[ \"weight\" ]\nX_valid = valid[ CFG.feature_cols ]\ny_valid = valid[ CFG.target_col ]\nw_valid = valid[ \"weight\" ]\n\n(X_train.shape, y_train.shape, w_train.shape, X_valid.shape, y_valid.shape, w_valid.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T23:55:46.916466Z","iopub.execute_input":"2025-05-27T23:55:46.916898Z","iopub.status.idle":"2025-05-27T23:56:34.730578Z","shell.execute_reply.started":"2025-05-27T23:55:46.916856Z","shell.execute_reply":"2025-05-27T23:56:34.729421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# make room\n\nimport gc\ndel train\ndel valid\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T23:56:34.732671Z","iopub.execute_input":"2025-05-27T23:56:34.733058Z","iopub.status.idle":"2025-05-27T23:56:35.026767Z","shell.execute_reply.started":"2025-05-27T23:56:34.733027Z","shell.execute_reply":"2025-05-27T23:56:35.025936Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Model Fit","metadata":{}},{"cell_type":"code","source":"# Scale the features\n\nscaler = MinMaxScaler()\nX_train = scaler.fit_transform(X_train.values)\nX_valid = scaler.fit_transform(X_valid.values)\ny_train = y_train.values\ny_valid = y_valid.values\n\n# ReShape the X & y\ndef create_sequences(input_x, input_y, time_steps):\n    X, y = [], [] # Note this will create large datasets\n    for i in range(len(input_x) - time_steps):\n        X.append(input_x[i:(i + time_steps)])\n        y.append(input_y[i + time_steps])\n    return np.array(X), np.array(y)\n\nX_train, y_train = create_sequences(X_train, y_train, 3)\nX_valid, y_valid = create_sequences(X_valid, y_valid, 3)\n\nX_train = X_train.reshape((X_train.shape[0], X_train.shape[1], X_train.shape[2]))\nX_valid = X_valid.reshape((X_valid.shape[0], X_valid.shape[1], X_valid.shape[2]))\n\n# Build the LSTM model\n\nmodel = Sequential()\nmodel.add(LSTM(\n    50, activation='relu', return_sequences=True, \n    input_shape=(X_train.shape[1], X_train.shape[2])\n))\nmodel.add(Dropout(0.2))\nmodel.add(LSTM(50, activation='relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1))  # Output layer for regression","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T23:56:35.028817Z","iopub.execute_input":"2025-05-27T23:56:35.029118Z","iopub.status.idle":"2025-05-27T23:57:04.555577Z","shell.execute_reply.started":"2025-05-27T23:56:35.029094Z","shell.execute_reply":"2025-05-27T23:57:04.554390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compile and fit the model\n# NOTE: It takes quite a long time\n\nmodel.compile(optimizer='adam', loss='mean_squared_error')\nmodel.fit(X_train, y_train, epochs=50, batch_size=32)\n\n# Model prediction & evaluation\n\ny_pred_train = model.predict(X_train)\ntrain_score = r2_score(y_train, y_pred_train, sample_weight=w_train)\n\ny_pred_valid = model.predict(X_valid)\nvalid_score = r2_score(y_valid, y_pred_valid, sample_weight=w_valid)\n\nprint(f\"Train R2: {train_score}, Validation R2: {valid_score}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-27T23:57:04.556379Z","iopub.execute_input":"2025-05-27T23:57:04.556659Z","execution_failed":"2025-05-28T12:29:20.508Z"}},"outputs":[],"execution_count":null}]}