{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":216512974,"sourceType":"kernelVersion"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This was my initial model creation notebook from the JS market prediction competition. I'm new to both polars and neural networks so this was a learning experience foremost. I took the approach of treating the symbol_id column as a categorical feature and feeding it into an embedding layer before passing it as an input to the rest of the model. Whilst results in training/validation were excellent, I spent too long wrestling with the submission API to implement cross validation and my \"long term model\" trained on moving averages to try to capture some of the longer timescale movements fell into the NaN loss pitfall. This could also be improved by adding handling for unknown symbol_ids, which I also believe may have impacted the submission score.","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport pickle\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import backend as K\nfrom keras import regularizers\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras import callbacks\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout, Embedding\nfrom sklearn.metrics import mean_squared_error, r2_score\nos.environ[\"CUDA_VISIBLE_DEVICES\"]='0,1'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T21:54:39.326245Z","iopub.execute_input":"2025-01-13T21:54:39.326539Z","iopub.status.idle":"2025-01-13T21:54:48.409164Z","shell.execute_reply.started":"2025-01-13T21:54:39.326505Z","shell.execute_reply":"2025-01-13T21:54:48.408178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.test.is_gpu_available()\ntf.random.set_seed(42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T21:54:48.410234Z","iopub.execute_input":"2025-01-13T21:54:48.410943Z","iopub.status.idle":"2025-01-13T21:54:48.417109Z","shell.execute_reply.started":"2025-01-13T21:54:48.410904Z","shell.execute_reply":"2025-01-13T21:54:48.416145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CONFIG:\n    target_col = \"responder_6\"\n    all_cols = [\"date_id\", \"symbol_id\", \"time_id\", \"weight\"] + [f\"feature_{idx:02d}\" for idx in range(79)]+ [f\"responder_{idx}_lag_1\" for idx in range(9)] + [target_col] + [f\"feature_{idx:02d}_mvg_avg\" for idx in range(79)]\n    feature_cols = [f\"feature_{idx:02d}\" for idx in range(79)]\n    rolling_cols = [f\"feature_{idx:02d}_mvg_avg\" for idx in range(79)]\n    lag_cols = [f\"responder_{idx}_lag_1\" for idx in range(9)] \n    timeseries_cols = [\"date_id\", \"time_id\", \"weight\"]\n    shorttermcols = [\"date_id\", \"symbol_id\", \"time_id\", \"weight\"] + [f\"feature_{idx:02d}\" for idx in range(79)]+ [f\"responder_{idx}_lag_1\" for idx in range(9)] + [target_col]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T21:54:48.418114Z","iopub.execute_input":"2025-01-13T21:54:48.418441Z","iopub.status.idle":"2025-01-13T21:54:48.431717Z","shell.execute_reply.started":"2025-01-13T21:54:48.418409Z","shell.execute_reply":"2025-01-13T21:54:48.430898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainfiles = glob.glob(\"/kaggle/input/js24-preprocessing-create-lags-and-rolling/training.parquet/date_id=*/00000000.parquet\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T21:54:48.433809Z","iopub.execute_input":"2025-01-13T21:54:48.434101Z","iopub.status.idle":"2025-01-13T21:54:50.836469Z","shell.execute_reply.started":"2025-01-13T21:54:48.434078Z","shell.execute_reply":"2025-01-13T21:54:50.835631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainfiles.sort()\ntrainfiles","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T21:54:50.837560Z","iopub.execute_input":"2025-01-13T21:54:50.837933Z","iopub.status.idle":"2025-01-13T21:54:50.851367Z","shell.execute_reply.started":"2025-01-13T21:54:50.837904Z","shell.execute_reply":"2025-01-13T21:54:50.850413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pl_train = pl.concat([pl.read_parquet(_f, columns=CONFIG.shorttermcols) for _f in trainfiles[350:]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T21:54:50.852222Z","iopub.execute_input":"2025-01-13T21:54:50.852498Z","iopub.status.idle":"2025-01-13T21:55:04.255469Z","shell.execute_reply.started":"2025-01-13T21:54:50.852466Z","shell.execute_reply":"2025-01-13T21:55:04.254717Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pl_train.schema","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T21:55:35.163930Z","iopub.execute_input":"2025-01-13T21:55:35.164266Z","iopub.status.idle":"2025-01-13T21:55:35.175266Z","shell.execute_reply.started":"2025-01-13T21:55:35.164241Z","shell.execute_reply":"2025-01-13T21:55:35.174426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pl_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:24.987974Z","iopub.execute_input":"2025-01-13T15:00:24.988202Z","iopub.status.idle":"2025-01-13T15:00:24.994193Z","shell.execute_reply.started":"2025-01-13T15:00:24.988181Z","shell.execute_reply":"2025-01-13T15:00:24.993367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pl_train = pl_train.fill_null(strategy=\"mean\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:24.996272Z","iopub.execute_input":"2025-01-13T15:00:24.996518Z","iopub.status.idle":"2025-01-13T15:00:25.293397Z","shell.execute_reply.started":"2025-01-13T15:00:24.996497Z","shell.execute_reply":"2025-01-13T15:00:25.292469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pl_train = pl_train.to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:25.294837Z","iopub.execute_input":"2025-01-13T15:00:25.295186Z","iopub.status.idle":"2025-01-13T15:00:30.051852Z","shell.execute_reply.started":"2025-01-13T15:00:25.295152Z","shell.execute_reply":"2025-01-13T15:00:30.051118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Y_train = pl_train[CONFIG.target_col]\nX_embedding = pl_train[\"symbol_id\"]\npl_train.drop([CONFIG.target_col, 'symbol_id'], axis=1, inplace=True)\ninput_dimension = X_embedding.max()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:30.052619Z","iopub.execute_input":"2025-01-13T15:00:30.052833Z","iopub.status.idle":"2025-01-13T15:00:31.075203Z","shell.execute_reply.started":"2025-01-13T15:00:30.052806Z","shell.execute_reply":"2025-01-13T15:00:31.074509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\n    \"Y \", Y_train.shape,\n    \"X embed \", X_embedding.shape,\n    \"train \", pl_train.shape\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:31.075965Z","iopub.execute_input":"2025-01-13T15:00:31.076185Z","iopub.status.idle":"2025-01-13T15:00:31.081571Z","shell.execute_reply.started":"2025-01-13T15:00:31.076167Z","shell.execute_reply":"2025-01-13T15:00:31.080789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pl_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:31.082180Z","iopub.execute_input":"2025-01-13T15:00:31.082374Z","iopub.status.idle":"2025-01-13T15:00:31.117380Z","shell.execute_reply.started":"2025-01-13T15:00:31.082357Z","shell.execute_reply":"2025-01-13T15:00:31.116777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"hidden_units = (128,64,32) #basic pyramid shape for initial attempt\nstock_embedding_size = 8 #dimensionality of output\n\n\ndef short_term_model():\n    \n    # Splitting the inputs to the symbol id for embedding and the rest of the numerical data\n    symbol_id_input = keras.Input(shape=(1,), name='symbol_id')\n    num_input = keras.Input(shape=(91,), name='num_data')\n\n\n    #embedding, flatenning and concatenating\n    symbol_embedded = keras.layers.Embedding(input_dimension+1, stock_embedding_size, \n                                           input_length=1, name='symbol_embedding')(symbol_id_input)\n    symbol_flattened = keras.layers.Flatten()(symbol_embedded)\n    out = keras.layers.Concatenate()([symbol_flattened, num_input])\n    \n    # Add one or more hidden layers\n    for n_hidden in hidden_units:\n\n        out = keras.layers.BatchNormalization()(out)\n        out = keras.layers.Dense(n_hidden, activation='swish', activity_regularizer=regularizers.l2(0.01))(out)\n        #out = keras.layers.Dropout(rate=0.02,seed=42)(out)\n        \n\n    #out = keras.layers.Concatenate()([out, num_input])\n\n    # A single output: our predicted rating\n    out = keras.layers.Dense(1, activation='linear', name='prediction')(out)\n    \n    model = keras.Model(\n    inputs = [symbol_id_input, num_input],\n    outputs = out,\n    )\n    \n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:31.118114Z","iopub.execute_input":"2025-01-13T15:00:31.118368Z","iopub.status.idle":"2025-01-13T15:00:31.124144Z","shell.execute_reply.started":"2025-01-13T15:00:31.118347Z","shell.execute_reply":"2025-01-13T15:00:31.123299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"short_term_model().summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:31.124913Z","iopub.execute_input":"2025-01-13T15:00:31.125100Z","iopub.status.idle":"2025-01-13T15:00:31.289767Z","shell.execute_reply.started":"2025-01-13T15:00:31.125084Z","shell.execute_reply":"2025-01-13T15:00:31.289110Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = short_term_model()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:31.290473Z","iopub.execute_input":"2025-01-13T15:00:31.290729Z","iopub.status.idle":"2025-01-13T15:00:31.343236Z","shell.execute_reply.started":"2025-01-13T15:00:31.290696Z","shell.execute_reply":"2025-01-13T15:00:31.342656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(\n    tf.optimizers.Adam(0.0001),  #starting point learning rate, noisy data so may as well go slow\n    loss='MSE',\n    metrics=['R2Score'],\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:31.343847Z","iopub.execute_input":"2025-01-13T15:00:31.344035Z","iopub.status.idle":"2025-01-13T15:00:31.358305Z","shell.execute_reply.started":"2025-01-13T15:00:31.344018Z","shell.execute_reply":"2025-01-13T15:00:31.357645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"early_stopping = callbacks.EarlyStopping(\n    min_delta=0.0005, # minimium amount of change to count as an improvement\n    patience=5, # how many epochs to wait before stopping\n    restore_best_weights=True,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:31.359320Z","iopub.execute_input":"2025-01-13T15:00:31.359629Z","iopub.status.idle":"2025-01-13T15:00:31.363151Z","shell.execute_reply.started":"2025-01-13T15:00:31.359598Z","shell.execute_reply":"2025-01-13T15:00:31.362226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fastpredictor = model.fit(\n[X_embedding, pl_train], #multiple input for embedded and numeric\nY_train, #target\nbatch_size=1500,\nepochs=2000,\nverbose=1,\ncallbacks=[early_stopping],\nvalidation_split=.1,\nshuffle = True,     \n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:00:31.363789Z","iopub.execute_input":"2025-01-13T15:00:31.364015Z","iopub.status.idle":"2025-01-13T15:04:56.035794Z","shell.execute_reply.started":"2025-01-13T15:00:31.363985Z","shell.execute_reply":"2025-01-13T15:04:56.034847Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_df = pd.DataFrame(fastpredictor.history)\nhistory_df.loc[:, ['loss', 'val_loss']].plot(ylim = (0, 1));\nprint(\"Minimum validation loss: {}\".format(history_df['val_loss'].min()))\nprint(\"Corresponding score: {}\".format(history_df.loc[history_df['val_loss'].idxmin()]['val_R2Score']\n))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:04:56.036911Z","iopub.execute_input":"2025-01-13T15:04:56.037157Z","iopub.status.idle":"2025-01-13T15:04:56.330012Z","shell.execute_reply.started":"2025-01-13T15:04:56.037136Z","shell.execute_reply":"2025-01-13T15:04:56.329307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output_path = '/kaggle/working/shortmodel.keras'\nmodel.save(output_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-13T15:04:56.330927Z","iopub.execute_input":"2025-01-13T15:04:56.331257Z","iopub.status.idle":"2025-01-13T15:04:56.398754Z","shell.execute_reply.started":"2025-01-13T15:04:56.331222Z","shell.execute_reply":"2025-01-13T15:04:56.398141Z"}},"outputs":[],"execution_count":null}]}