{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"},{"sourceId":217587,"sourceType":"modelInstanceVersion","modelInstanceId":185551,"modelId":207681}],"dockerImageVersionId":30823,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:26.322894Z","iopub.execute_input":"2025-01-10T03:31:26.323219Z","iopub.status.idle":"2025-01-10T03:31:26.662431Z","shell.execute_reply.started":"2025-01-10T03:31:26.323192Z","shell.execute_reply":"2025-01-10T03:31:26.661607Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport polars as pl\nimport gc\n\ntest_df = pl.scan_parquet(\n    f\"/kaggle/input/preprocessed_data_mini/tensorflow2/default/1/training_data.parquet\"\n).collect().to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:28.571556Z","iopub.execute_input":"2025-01-10T03:31:28.571981Z","iopub.status.idle":"2025-01-10T03:31:43.405177Z","shell.execute_reply.started":"2025-01-10T03:31:28.571953Z","shell.execute_reply":"2025-01-10T03:31:43.404256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_df = pl.scan_parquet(\n    f\"/kaggle/input/preprocessed_data_mini/tensorflow2/default/1/validation_data.parquet\"\n).collect().to_pandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:43.406453Z","iopub.execute_input":"2025-01-10T03:31:43.406723Z","iopub.status.idle":"2025-01-10T03:31:44.014218Z","shell.execute_reply.started":"2025-01-10T03:31:43.406699Z","shell.execute_reply":"2025-01-10T03:31:44.013526Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Calling preprocessed data model","metadata":{}},{"cell_type":"code","source":"# 데이터 LSTM 입력화\n\ndef dat_to_input(data):\n    features = data.filter(regex='^feature_')\n    responders = data.filter(regex='^responder_')\n    weights = data['weight']\n\n    gc.collect()\n\n    # Convert to numpy arrays for TensorFlow\n    X = features.values  # Features for input\n\n    # Assuming you have a DataFrame `y_train` with all responders <- noooo\n    y = responders[['responder_6']].values  # Keep only responder_6\n\n    # run if nan or inf exists - \n    #X = np.nan_to_num(X, nan=0.0, posinf=0.0, neginf=0.0)\n    #y = np.nan_to_num(y, nan=0.0, posinf=0.0, neginf=0.0)\n\n    return X,y\n\n\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:44.015486Z","iopub.execute_input":"2025-01-10T03:31:44.015736Z","iopub.status.idle":"2025-01-10T03:31:44.06583Z","shell.execute_reply.started":"2025-01-10T03:31:44.015701Z","shell.execute_reply":"2025-01-10T03:31:44.064918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X, y = dat_to_input(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:44.067097Z","iopub.execute_input":"2025-01-10T03:31:44.067344Z","iopub.status.idle":"2025-01-10T03:31:46.789455Z","shell.execute_reply.started":"2025-01-10T03:31:44.067324Z","shell.execute_reply":"2025-01-10T03:31:46.788723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_val, y_val = dat_to_input(val_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:46.790302Z","iopub.execute_input":"2025-01-10T03:31:46.790542Z","iopub.status.idle":"2025-01-10T03:31:46.976953Z","shell.execute_reply.started":"2025-01-10T03:31:46.790521Z","shell.execute_reply":"2025-01-10T03:31:46.976039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:46.977796Z","iopub.execute_input":"2025-01-10T03:31:46.97803Z","iopub.status.idle":"2025-01-10T03:31:46.982643Z","shell.execute_reply.started":"2025-01-10T03:31:46.97801Z","shell.execute_reply":"2025-01-10T03:31:46.981979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = X.reshape(-1, 3, 79)\nX_val = X_val.reshape(-1,1,79)\n\nprint(X.shape, X_val.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:46.983363Z","iopub.execute_input":"2025-01-10T03:31:46.983603Z","iopub.status.idle":"2025-01-10T03:31:46.997233Z","shell.execute_reply.started":"2025-01-10T03:31:46.983583Z","shell.execute_reply":"2025-01-10T03:31:46.996403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:46.999052Z","iopub.execute_input":"2025-01-10T03:31:46.999278Z","iopub.status.idle":"2025-01-10T03:31:47.059679Z","shell.execute_reply.started":"2025-01-10T03:31:46.99926Z","shell.execute_reply":"2025-01-10T03:31:47.058928Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# AutoEncoder","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import MinMaxScaler\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Dense, LSTM, GRU, Input, TimeDistributed, Reshape\nfrom keras.layers import GRU, Dense, Dropout, BatchNormalization\nfrom sklearn.preprocessing import MinMaxScaler","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:47.060576Z","iopub.execute_input":"2025-01-10T03:31:47.060836Z","iopub.status.idle":"2025-01-10T03:31:54.479051Z","shell.execute_reply.started":"2025-01-10T03:31:47.060809Z","shell.execute_reply":"2025-01-10T03:31:54.478362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.losses import Huber\nfrom keras.optimizers import Adam\n\n# 입력 레이어\ninput_layer = Input(shape=(X.shape[1], X.shape[2]))  # (타임스텝, 특성)\n\n# 인코더: 다층 LSTM\nencoded = LSTM(128, activation='relu', return_sequences=True)(input_layer)  # 첫 번째 LSTM\nencoded = Dropout(0.2)(encoded)  # Dropout으로 과적합 방지\nencoded = LSTM(64, activation='relu', return_sequences=False)(encoded)  # 두 번째 LSTM\nencoded = BatchNormalization()(encoded)  # BatchNormalization으로 학습 안정화\n\n# Bottleneck\nbottleneck = Dense(64, activation='relu')(encoded)  # 잠재 공간\nbottleneck = Dropout(0.2)(bottleneck)  # Dropout 추가\n\n# 디코더: Dense -> LSTM\ndecoded = Dense(X.shape[1] * X.shape[2], activation='relu')(bottleneck)  # Latent 공간을 펼치기\ndecoded = Reshape((X.shape[1], X.shape[2]))(decoded)  # 원래 시계열 차원으로 복원\ndecoded = LSTM(64, activation='relu', return_sequences=True)(decoded)  # LSTM으로 시계열 복원\ndecoded = Dropout(0.2)(decoded)\ndecoded = LSTM(X.shape[2], activation='relu', return_sequences=True)(decoded)\n\n# Autoencoder 모델\nautoencoder = Model(inputs=input_layer, outputs=decoded)\nautoencoder.compile(optimizer=Adam(learning_rate=0.001), loss=Huber(delta=1.0))  # Huber Loss 적용\nautoencoder.summary()\n\n# 학습\nautoencoder.fit(X, X, epochs=5, batch_size=64, validation_split=0.2, verbose=1)  # Epoch 조정 가능\n\n# 인코더 모델\nencoder = Model(inputs=input_layer, outputs=bottleneck)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:31:54.479794Z","iopub.execute_input":"2025-01-10T03:31:54.480252Z","iopub.status.idle":"2025-01-10T03:43:59.437806Z","shell.execute_reply.started":"2025-01-10T03:31:54.48023Z","shell.execute_reply":"2025-01-10T03:43:59.437089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:43:59.438588Z","iopub.execute_input":"2025-01-10T03:43:59.438827Z","iopub.status.idle":"2025-01-10T03:43:59.659514Z","shell.execute_reply.started":"2025-01-10T03:43:59.438806Z","shell.execute_reply":"2025-01-10T03:43:59.658715Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n# Autoencoder 모델 구축\ninput_layer = Input(shape=(X.shape[1], X.shape[2]))  # (타임스텝, 특성)\nencoded = LSTM(128, activation='relu', return_sequences=False)(input_layer)  # 인코더\n\nbottleneck = Dense(64, activation='relu')(encoded)  # 반환점 - bottleneck\n\n# 디코더: Latent 공간 -> 원래 시계열 차원 복원\ndecoded = Dense(X.shape[1] * X.shape[2], activation='relu')(bottleneck)  # 전체 차원 펼치기\ndecoded = Reshape((X.shape[1], X.shape[2]))(decoded)  # (타임스텝, 특성)으로 복원\n\nautoencoder = Model(inputs=input_layer, outputs=decoded)\nautoencoder.compile(optimizer='adam', loss='mse')\nautoencoder.summary()\n\n# Autoencoder 학습\nautoencoder.fit(X, X, epochs=5, batch_size=32, validation_split=0.2, verbose=1) # 여기를 조정해서 학습 시간 단축 가능 - 현재 20분 정도\n\n# Encoder 모델 구축\nencoder = Model(inputs=input_layer, outputs=bottleneck)\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:43:59.660262Z","iopub.execute_input":"2025-01-10T03:43:59.660481Z","iopub.status.idle":"2025-01-10T03:43:59.674499Z","shell.execute_reply.started":"2025-01-10T03:43:59.660461Z","shell.execute_reply":"2025-01-10T03:43:59.673702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\n# 3. GRU 모델 구축\n# Autoencoder를 통해 Latent Representation 추출\nlatent_train = encoder.predict(X)\nlatent_test = encoder.predict(X_val)\n\n# GRU를 활용한 예측\ngru_model = Sequential([\n    GRU(128, activation='relu', input_shape=(1, latent_train.shape[1]), return_sequences=False),\n    Dense(64, activation='relu'),\n    Dense(y.shape[1], activation='linear')  # 다중 특성 예측\n])\n\ngru_model.compile(optimizer='adam', loss='mse', metrics=['mae'])\ngru_model.summary()\n\n# GRU 입력 차원 맞추기\nlatent_train = latent_train.reshape(latent_train.shape[0], 1, latent_train.shape[1])\nlatent_test = latent_test.reshape(latent_test.shape[0], 1, latent_test.shape[1])\n\n# GRU 학습\ngru_model.fit(latent_train, y, epochs=5, batch_size=32, validation_split=0.2, verbose=1)\n\n# GRU 예측\npredictions = gru_model.predict(latent_test) # 다 같은 답을 내놓는다. 뭔가 문제가 있음. \n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:43:59.675271Z","iopub.execute_input":"2025-01-10T03:43:59.675552Z","iopub.status.idle":"2025-01-10T03:43:59.694541Z","shell.execute_reply.started":"2025-01-10T03:43:59.675522Z","shell.execute_reply":"2025-01-10T03:43:59.69375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nlatent_train = encoder.predict(X)\n\n# 1. GRU 모델 정의 및 학습\n# latent_train은 훈련 데이터의 특성 벡터, y_train은 목표 변수입니다.\ngru_model = Sequential([\n    GRU(128, activation='relu', input_shape=(1, latent_train.shape[1]), return_sequences=False),\n    Dense(64, activation='relu'),\n    Dense(latent_train.shape[1], activation='linear')  # y_train의 특성 수에 맞게\n])\n\ngru_model.compile(optimizer='adam', loss='mse', metrics=['mae'])\ngru_model.summary()\n'''\n# 아래 모델로 대체 가능? 아래 모델도 층이 그다지 깊어 보이지는 않는다...","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:43:59.695382Z","iopub.execute_input":"2025-01-10T03:43:59.695703Z","iopub.status.idle":"2025-01-10T03:43:59.70941Z","shell.execute_reply.started":"2025-01-10T03:43:59.695665Z","shell.execute_reply":"2025-01-10T03:43:59.70872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\ndef r_squared(y_true, y_pred):\n    \"\"\"\n    R-squared 계산 함수\n    y_true: 실제 값\n    y_pred: 예측 값\n    \"\"\"\n    # Residual sum of squares\n    ss_res = tf.reduce_sum(tf.square(y_true - y_pred))\n    \n    # Total sum of squares\n    ss_tot = tf.reduce_sum(tf.square(y_true - tf.reduce_mean(y_true)))\n    \n    # R-squared 계산\n    r2 = 1 - ss_res / (ss_tot + tf.keras.backend.epsilon())  # epsilon으로 0 나눔 방지\n    \n    return r2  # 반환값이 Tensor여야 함 - Tensor 함수 특징? ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T03:43:59.710209Z","iopub.execute_input":"2025-01-10T03:43:59.710539Z","iopub.status.idle":"2025-01-10T03:43:59.726328Z","shell.execute_reply.started":"2025-01-10T03:43:59.710508Z","shell.execute_reply":"2025-01-10T03:43:59.7256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.callbacks import EarlyStopping\n\n# X 훈련 데이터 autoencoder 학습시키기 및 MinMax로 스케일링\nscaler = MinMaxScaler()\n\nlatent_train = encoder.predict(X)\nlatent_train = scaler.fit_transform(latent_train)\n\ny = scaler.fit_transform(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T04:30:53.038931Z","iopub.execute_input":"2025-01-10T04:30:53.039303Z","iopub.status.idle":"2025-01-10T04:33:42.90664Z","shell.execute_reply.started":"2025-01-10T04:30:53.039274Z","shell.execute_reply":"2025-01-10T04:33:42.905683Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# GRU","metadata":{}},{"cell_type":"code","source":"# 1. GRU 모델 정의\ngru_model = Sequential([\n    # 첫 번째 GRU 레이어: 더 많은 유닛 추가 및 정규화 적용\n    GRU(128, activation='tanh', input_shape=(1, latent_train.shape[1]), return_sequences=True),\n    BatchNormalization(),\n    Dropout(0.3),  # 과적합 방지를 위한 드롭아웃\n\n    # 두 번째 GRU 레이어: 추가 유닛 및 정규화\n    GRU(64, activation='tanh', return_sequences=False),\n    BatchNormalization(),\n    Dropout(0.3),\n\n    # 첫 번째 Dense 레이어: 더 많은 노드 추가 및 활성화 함수 사용\n    Dense(64, activation='relu'), # 여기서 Relu를 통해 음수를 0으로 만든다. \n    BatchNormalization(),\n\n    # 출력 레이어: y_train의 특성 수에 맞는 선형 활성화 함수 사용\n    Dense(latent_train.shape[1], activation='linear')\n])\n\n# Early Stopping 추가\nearly_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n# 2. 컴파일 단계\ngru_model.compile(optimizer='adam', loss='mse', metrics=['mae'])\n\n# 3. 모델 구조 요약\ngru_model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T04:33:42.910728Z","iopub.execute_input":"2025-01-10T04:33:42.910961Z","iopub.status.idle":"2025-01-10T04:33:43.018076Z","shell.execute_reply.started":"2025-01-10T04:33:42.910942Z","shell.execute_reply":"2025-01-10T04:33:43.017252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T04:33:43.019394Z","iopub.execute_input":"2025-01-10T04:33:43.019723Z","iopub.status.idle":"2025-01-10T04:33:43.283305Z","shell.execute_reply.started":"2025-01-10T04:33:43.019692Z","shell.execute_reply":"2025-01-10T04:33:43.282331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# GRU 모델 훈련\n# latent_train을 3D 배열로 변환 (배치 크기, 타임스텝, 특성)\nlatent_train_reshaped = latent_train.reshape(latent_train.shape[0]//13, 13, latent_train.shape[1]) # Timestep을 3개에서 13개로 변경함. 필요하다면 다시 13 대신 다른 숫자로 변경이 가능함. \n\ngru_model_trained = gru_model.fit(latent_train_reshaped ,y ,epochs=50, batch_size=128, validation_split=0.2, callbacks=[early_stopping], verbose=1)\n\n# latent_train_reshaped가 훈련 데이터, y가 출력 데이터를 의미한다. 즉, y는 responder_6, latent_test는 디코더에 y를 던져 넣은 것.\n# 7정도에서 earlystopping이 작동한다. ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T04:34:36.947903Z","iopub.execute_input":"2025-01-10T04:34:36.948284Z","iopub.status.idle":"2025-01-10T04:38:23.208962Z","shell.execute_reply.started":"2025-01-10T04:34:36.948253Z","shell.execute_reply":"2025-01-10T04:38:23.208213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gru_model_trained.history\n\n# 모델 개선 방안 강구 ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T04:39:28.154365Z","iopub.execute_input":"2025-01-10T04:39:28.154698Z","iopub.status.idle":"2025-01-10T04:39:28.161556Z","shell.execute_reply.started":"2025-01-10T04:39:28.154669Z","shell.execute_reply":"2025-01-10T04:39:28.160465Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# RES Submit - fail to understand :( \n- guess I need more study","metadata":{}},{"cell_type":"code","source":"# Submit ERROR?? - CFG\n'''\nlags_ : pl.DataFrame | None = None\n\n\n# Replace this function with your inference code.\n# You can return either a Pandas or Polars dataframe, though Polars is recommended.\n# Each batch of predictions (except the very first) must be returned within 10 minutes of the batch features being provided.\ndef predict(test: pl.DataFrame, lags: pl.DataFrame | None) -> pl.DataFrame | pd.DataFrame:\n    \"\"\"Make a prediction.\"\"\"\n    # All the responders from the previous day are passed in at time_id == 0. We save them in a global variable for access at every time_id.\n    # Use them as extra features, if you like.\n    global lags_\n    if lags is not None:\n        lags_ = lags\n    # 1. Select the required feature columns and convert to numpy array for Keras\n    X_test = test.select(CFG.feature_columns).to_numpy()\n    X_test = np.where(np.isnan(X_test), mean_values, X_test)\n    X_test = (X_test - min_values) / max_min_diff\n    # 2. Make predictions using the Keras model\n    y_pred = gru_model_trained.predict(X_test, batch_size=4096)\n    \n    # 3. Prepare the DataFrame for output\n    predictions = test.select('row_id').with_columns(\n        pl.Series(\"responder_6\", y_pred.flatten())\n    )\n    # The predict function must return a DataFrame\n    assert isinstance(predictions, pl.DataFrame | pd.DataFrame)\n    # with columns 'row_id', 'responer_6'\n    assert predictions.columns == ['row_id', 'responder_6']\n    # and as many rows as the test data.\n    assert len(predictions) == len(test)\n\n    return predictions\n    '''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T04:40:45.303284Z","iopub.execute_input":"2025-01-10T04:40:45.303592Z","iopub.status.idle":"2025-01-10T04:40:45.309637Z","shell.execute_reply.started":"2025-01-10T04:40:45.303569Z","shell.execute_reply":"2025-01-10T04:40:45.308831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nimport os\nimport kaggle_evaluation.jane_street_inference_server\n\ninference_server = kaggle_evaluation.jane_street_inference_server.JSInferenceServer(predict)\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        (\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet',\n            '/kaggle/input/jane-street-real-time-market-data-forecasting/lags.parquet',\n        )\n    )\n'''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-10T04:41:45.047541Z","iopub.execute_input":"2025-01-10T04:41:45.04787Z","iopub.status.idle":"2025-01-10T04:41:45.095346Z","shell.execute_reply.started":"2025-01-10T04:41:45.047842Z","shell.execute_reply":"2025-01-10T04:41:45.094191Z"}},"outputs":[],"execution_count":null}]}