{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":12544477,"sourceType":"datasetVersion","datasetId":7919981}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_parquet('/kaggle/input/crypto-market-data/feature_agg')\ntrain_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = test_df.rename(columns={'prediction':'label'})\ntest_df.head()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import dask.dataframe as dd\nfrom sklearn.preprocessing import MinMaxScaler\nimport gc\n\n\ndf = pd.concat([train_df, test_df]).reset_index(drop=True)\ndask_df = dd.from_pandas(df, npartitions=100)\ndask_df.to_parquet(\"my_data.parquet\", compression=\"snappy\")\n\ndel test_df, train_df, df,\ngc.collect #free up memory\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nX_features = dd.read_parquet(\"my_data.parquet\").compute()\n\nprint(X_features.head())\nX_features.drop('label', axis=1, inplace=True)\nX_clean = X_features.astype(np.float32)\nprint(X_clean.shape)\ndel X_features\ngc.collect\nclean_dask = dd.from_pandas(X_clean, npartitions=100)\ndel X_clean\ngc.collect\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nscaler = MinMaxScaler()\n\nsample = clean_dask.sample(frac=0.01).compute()\nscaler = MinMaxScaler().fit(sample)\n\n# Define function to scale each partition\ndef scale_partition(partition):\n    return pd.DataFrame(scaler.transform(partition), columns=partition.columns)\n\nX_scaled = clean_dask.map_partitions(scale_partition)\n\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_scaled = X_scaled.compute()\nprint(X_scaled.shape)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del clean_dask\ngc.collect","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\nshutil.rmtree(\"./pca_chunks\", ignore_errors=True)\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.decomposition import IncrementalPCA\nimport matplotlib.pyplot as plt\n\n\n\nprint(\"3. DIMENSIONALITY REDUCTION\")\nprint(\"=\"*50)\n\nchunk_size = 1000\nipca = IncrementalPCA(n_components=100)\nfor i in range(0, X_scaled.shape[0], chunk_size):\n    chunk = X_scaled[i:i+chunk_size]\n    X_pca = ipca.fit_transform(chunk)\n    temp_df = pd.DataFrame(X_pca, columns=[f'pc{i}' for i in range(X_pca.shape[1])])\n    dd_chunk = dd.from_pandas(temp_df, npartitions=1)\n    \n    # Save each chunk into the same directory\n    dd_chunk.to_parquet(\"pca_chunks/\", append=True, ignore_divisions=True)\n\n# 6. Explained variance analysis\nexplained_variance_ratio = ipca.explained_variance_ratio_\ncumsum_variance = np.cumsum(explained_variance_ratio)\nn_components_90 = np.argmax(cumsum_variance >= 0.90) + 1\nn_components_95 = np.argmax(cumsum_variance >= 0.95) + 1\n\nprint(f\"Number of components needed for 90% variance: {n_components_90}\")\nprint(f\"Number of components needed for 95% variance: {n_components_95}\")\n\n# 7. Plot explained variance\nplt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nplt.plot(range(1, len(explained_variance_ratio) + 1),\n         explained_variance_ratio, 'bo-')\nplt.xlabel('Principal Component')\nplt.ylabel('Explained Variance Ratio')\nplt.title('PCA - Individual Explained Variance')\nplt.grid(True)\n\nplt.subplot(1, 2, 2)\nplt.plot(range(1, len(cumsum_variance) + 1),\n         cumsum_variance, 'ro-')\nplt.axhline(y=0.90, color='g', linestyle='--', label='90% variance')\nplt.axhline(y=0.95, color='b', linestyle='--', label='95% variance')\nplt.xlabel('Number of Components')\nplt.ylabel('Cumulative Explained Variance')\nplt.title('PCA - Cumulative Explained Variance')\nplt.legend()\nplt.grid(True)\n\nplt.tight_layout()\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del X_scaled, temp_df, X_pca, dd_chunk, chunk\ngc.collect","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pca_df = dd.read_parquet('pca_chunks/').compute()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.cluster import KMeans\n\n\nprint(\"\\n\" + \"=\"*50)\nprint(\"5. CLUSTERING ANALYSIS\")\nprint(\"=\"*50)\n\n\n# K-means clustering on PCA-reduced data\nprint(\"Performing K-means clustering...\")\nn_clusters_range = range(2, 11)\ninertias = []\n\nfor k in n_clusters_range:\n    kmeans = KMeans(n_clusters=k, random_state=42)\n    kmeans.fit(pca_df)  # Use first 50 PCA components\n    inertias.append(kmeans.inertia_)\n\n# Plot elbow curve\nplt.figure(figsize=(10, 6))\nplt.plot(n_clusters_range, inertias, 'bo-')\nplt.xlabel('Number of Clusters')\nplt.ylabel('Inertia')\nplt.title('K-means Clustering - Elbow Method')\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_aligned = dd.read_parquet('my_data.parquet', columns=['label']).compute()\ny_aligned.to_parquet('Y.parquet')","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\noptimal_k = 4  # You can adjust based on elbow curve\nkmeans_final = KMeans(n_clusters=optimal_k, random_state=42)\nclusters = kmeans_final.fit_predict(pca_df)\n# Analyze clusters\ncluster_analysis = pd.DataFrame({\n    'cluster': clusters,\n    'target': y_aligned.squeeze()\n})\n\nprint(f\"\\nCluster analysis with {optimal_k} clusters:\")\ncluster_stats = cluster_analysis.groupby('cluster')['target'].agg(['count', 'mean', 'std'])\nprint(cluster_stats)\n\n# Visualize clusters using first 2 PCA components\nplt.figure(figsize=(12, 5))\n\nplt.subplot(1, 2, 1)\nscatter = plt.scatter(pca_df.iloc[:, 0], pca_df.iloc[:, 1], c=clusters, alpha=0.6, cmap='tab10')\nplt.colorbar(scatter, label='Cluster')\nplt.xlabel('First Principal Component')\nplt.ylabel('Second Principal Component')\nplt.title('K-means Clusters in PCA Space')\n\nplt.subplot(1, 2, 2)\nscatter = plt.scatter(pca_df.iloc[:, 0], pca_df.iloc[:, 1], c=y_aligned.squeeze(), alpha=0.6, cmap='viridis')\nplt.colorbar(scatter, label='Target Value')\nplt.xlabel('First Principal Component')\nplt.ylabel('Second Principal Component')\nplt.title('Target Values in PCA Space')\n\nplt.tight_layout()\nplt.show()\n\n\n","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del y_aligned\ngc.collect","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#add distance to cluster centroid as a feature\nfrom scipy.stats import pearsonr\n\nfrom scipy.spatial.distance import cdist\ndistances = cdist(pca_df, kmeans_final.cluster_centers_, metric='euclidean')  # shape: (n_samples, n_clusters)\nfor i in range(distances.shape[1]):\n    pca_df[f'distance_to_centroid_{i}'] = distances[:, i]\nmean_distances = distances.mean(axis=1)  # shape: (n_samples,)\npca_df['mean_distance_to_centroids'] = mean_distances\npca_df['cluster_label'] = clusters\npca_df = pd.get_dummies(pca_df, columns=['cluster_label'], prefix='cluster')\npca_df.to_parquet('pca_df.parquet')","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\"\"\"#Import required libraries\nimport math\nimport time\nimport datetime\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport optuna\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import StandardScaler\nfrom keras.models import save_model\nfrom keras.layers import SimpleRNN, LSTM, GRU, Bidirectional, Dense, Dropout\nfrom keras.layers import MultiHeadAttention, LayerNormalization, Input, GlobalAveragePooling1D, Embedding, Add\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom statsmodels.tsa.holtwinters import ExponentialSmoothing\nfrom keras.optimizers import Adam\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score, explained_variance_score\nfrom keras.models import Sequential, Model\nfrom hyperopt import fmin, tpe, hp, Trials, STATUS_OK, STATUS_FAIL, space_eval\nimport tensorflow as tf\nfrom keras.models import load_model\nimport os\nfrom tensorflow.keras import backend as K\nfrom optuna.integration import TFKerasPruningCallback\n\n\n# Apply Holt-Winters Exponential Smoothing with multiplicative trend and seasonality\nhw_model = ExponentialSmoothing(close_prices, trend='mul', seasonal='mul', seasonal_periods=365)\n\nhw_fit = hw_model.fit()\n\n# View optimized parameters\noptimized_params = hw_fit.params\n\n# Extract level, trend, and seasonal components\ndf['HW_Level'] = hw_fit.level\ndf['HW_Trend'] = hw_fit.trend\ndf['HW_Seasonal'] = hw_fit.season\n\n# Deseasonalize the data\ndf['Deseasonalized'] = df['Close'] / (df['HW_Seasonal'] * df['HW_Level'])\n\n# Length of training data\nvalues = df['Deseasonalized'].values\ntraining_data_len = math.ceil(len(values) * 0.80)\n\ntime_step = 30\n\n# Split data into training and test sets before scaling\ntrain_values = values[:training_data_len]\n\ntest_values = values[training_data_len - time_step:]\n\n# Normalize using only training data\nscaler = MinMaxScaler(feature_range=(0, 1))\nscaled_train = scaler.fit_transform(train_values.reshape(-1, 1))\nscaled_test = scaler.transform(test_values.reshape(-1, 1))\n\n# Build training sequences\nx_train, y_train = [], []\n\nfor i in range(time_step, len(scaled_train)):\n    x_train.append(scaled_train[i - time_step:i, 0])\n    y_train.append(scaled_train[i, 0])\n\nx_train = np.array(x_train)\ny_train = np.array(y_train)\n\n# Reshape x_train to have the shape (samples, time_step, 1) for the transformer model\nx_train = np.reshape(x_train, (x_train.shape[0], x_train.shape[1], 1))\n\n# Build test sequences\nx_test = []\n\nfor i in range(time_step, len(scaled_test)):\n    x_test.append(scaled_test[i - time_step:i, 0])\n\nx_test = np.array(x_test)\n\n# Reshape x_test to have the shape (samples, time_step, 1)\nx_test = np.reshape(x_test, (x_test.shape[0], x_test.shape[1], 1))\n\n# y_test is taken from the original close_prices from the start of the test period onward\ny_test = close_prices.values[training_data_len:]\n\n# Define the Mish activation function\ndef mish(x):\n    return x * K.tanh(K.softplus(x))\n\n# Define the objective function for Optuna\ndef objective(trial):\n    # Hyperparameters to tune\n    learning_rate = trial.suggest_float('learning_rate', 0.0001, 0.01, log=True)\n    units = trial.suggest_int('units', 20, 50, step=5)\n    dropout_rate = trial.suggest_float('dropout_rate', 0, 0.3)\n    batch_size = trial.suggest_categorical('batch_size', [16, 32, 64, 128])\n    epochs = trial.suggest_int('epochs', 50, 150, step=10)\n    num_blocks = trial.suggest_int('num_blocks', 1, 4)\n    num_heads = trial.suggest_int('num_heads', 2, 10, step=2)\n    head_size = trial.suggest_int('head_size', 8, 64, step=8)\n\n    # Define the model architecture\n    def create_helformer_model(input_shape):\n        inputs = Input(shape=input_shape)\n        x = inputs\n\n        for _ in range(num_blocks):\n            x_norm1 = LayerNormalization(epsilon=1e-6)(x)\n            attention_output = MultiHeadAttention(num_heads=num_heads, key_dim=head_size, dropout=dropout_rate)(x_norm1, x_norm1)\n            x = x + attention_output\n\n            x_norm2 = LayerNormalization(epsilon=1e-6)(x)\n            x = x + x_norm2\n\n        # LSTM layer\n        x = LSTM(units, activation=mish, return_sequences=False)(x)\n\n        # Output layer\n        outputs = Dense(1)(x)\n\n        model = Model(inputs=inputs, outputs=outputs)\n        return model\n\n    # Create and compile the model\n    input_shape = (x_train.shape[1], x_train.shape[2])\n    model = create_helformer_model(input_shape)\n    model.compile(optimizer=Adam(learning_rate=learning_rate), loss='mean_squared_error')\n\n    # Train the model\n    history = model.fit(\n        x_train, y_train,\n        batch_size=batch_size,\n        epochs=epochs,\n        validation_split=0.2,\n        verbose=0,\n        callbacks=[TFKerasPruningCallback(trial, 'val_loss')]\n    )\n\n    # Predictions on validation data\n    val_loss = min(history.history['val_loss'])\n    return val_loss\n\n# Run the Optuna optimization\nstudy = optuna.create_study(direction='minimize')\nstudy.optimize(objective, n_trials=50)\n\n# Print the best hyperparameters\nbest_params = study.best_params\nprint(\"Best hyperparameters:\", best_params)\n\n# Retrain the model using the best hyperparameters\ndef create_best_helformer_model(input_shape):\n    inputs = Input(shape=input_shape)\n    x = inputs\n\n    for _ in range(best_params['num_blocks']):\n        x_norm1 = LayerNormalization(epsilon=1e-6)(x)\n        attention_output = MultiHeadAttention(num_heads=best_params['num_heads'], key_dim=best_params['head_size'], dropout=best_params['dropout_rate'])(x_norm1, x_norm1)\n        x = x + attention_output\n\n        x_norm2 = LayerNormalization(epsilon=1e-6)(x)\n        x = x + x_norm2\n\n    # LSTM layer\n    x = LSTM(best_params['units'], activation=mish, return_sequences=False)(x)\n\n    # Output layer\n    outputs = Dense(1)(x)\n\n    model = Model(inputs=inputs, outputs=outputs)\n    return model\n\n# Create and compile the model with the best parameters\nfinal_model_Helformer = create_best_helformer_model((x_train.shape[1], x_train.shape[2]))\nfinal_model_Helformer.compile(optimizer=Adam(learning_rate=best_params['learning_rate']), loss='mean_squared_error')\n\n# Train the model\nprint(\"Training final Helformer model\")\nfinal_history_Helformer = final_model_Helformer.fit(x_train, y_train, batch_size=best_params['batch_size'], epochs=best_params['epochs'], verbose=0)# Verbose set to 1 for detailed output during training\n\nfinal_model_Helformer.save('Main_Helformer.h5')\n\n# Predictions on training data\ntrain_predictions = final_model_Helformer.predict(x_train)\ntrain_predictions = scaler.inverse_transform(train_predictions).flatten()\n\n# Restore level and seasonal components for training predictions\ntrain_predictions = train_predictions * df['HW_Seasonal'].values[time_step:training_data_len] * df['HW_Level'].values[time_step:training_data_len]\n\ny_train_rescaled = scaler.inverse_transform([y_train]).flatten()\ny_train_rescaled = y_train_rescaled * df['HW_Seasonal'].values[time_step:training_data_len] * df['HW_Level'].values[time_step:training_data_len]\n\n# Function to calculate MAPE\ndef mean_absolute_percentage_error(y_true, y_pred):\n    return np.mean(np.abs((y_true - y_pred) / y_true)) * 100\n\n# Function to calculate KGE (Kling-Gupta Efficiency)\ndef kge(y_true, y_pred):\n    cc = np.corrcoef(y_true, y_pred)[0, 1]  # Correlation coefficient\n    alpha = np.std(y_pred) / np.std(y_true)  # Variability ratio\n    beta = np.mean(y_pred) / np.mean(y_true)  # Bias ratio\n    return 1 - np.sqrt((cc - 1) ** 2 + (alpha - 1) ** 2 + (beta - 1) ** 2)\n\n# Predictions on testing data\nprint(\"Predicting test data\")\ntest_predictions = final_model_Helformer.predict(x_test)\ntest_predictions = scaler.inverse_transform(test_predictions)\ntest_predictions = test_predictions.flatten()\n\n# Restore level and seasonal components for test predictions\ntest_predictions = test_predictions * df['HW_Seasonal'].values[training_data_len:] * df['HW_Level'].values[training_data_len:]\n\n# Calculate metrics for training data\ntrain_mse = mean_squared_error(y_train_rescaled, train_predictions)\ntrain_rmse = np.sqrt(train_mse)\ntrain_mape = mean_absolute_percentage_error(y_train_rescaled, train_predictions)\ntrain_mae = mean_absolute_error(y_train_rescaled, train_predictions)\ntrain_r2 = r2_score(y_train_rescaled, train_predictions)\ntrain_evs = explained_variance_score(y_train_rescaled, train_predictions)\ntrain_kge = kge(y_train_rescaled, train_predictions)\n\n# Calculate metrics for testing data\ntest_mse = mean_squared_error(y_test, test_predictions)\ntest_rmse = np.sqrt(test_mse)\ntest_mape = mean_absolute_percentage_error(y_test, test_predictions)\ntest_mae = mean_absolute_error(y_test, test_predictions)\ntest_r2 = r2_score(y_test, test_predictions)\ntest_evs = explained_variance_score(y_test, test_predictions)\ntest_kge = kge(y_test, test_predictions)\n\n# Get current date and time\nnow = datetime.datetime.now()\ntimestamp = now.strftime(\"%Y-%m-%d_%H-%M-%S\")\nreadable_time = now.strftime(\"%Y-%m-%d %H:%M:%S\")\n\n# Define file path and name\nfilename = f\"Performance_metrics_{timestamp}.txt\"\nfilepath = os.path.join(\"/results/\", filename)\n\n\n# Write metrics to file\nwith open(filepath, 'w') as f:               # Change filename to filepath for codeocean\n    f.write(f\"::     ::\\n\")\n    f.write(f\"Training Data Evaluation Metrics\\n\")\n    f.write(f\"Timestamp: {readable_time}\\n\\n\")\n    f.write(f\"MSE: {train_mse:.4f}\\n\")\n    f.write(f\"RMSE: {train_rmse:.4f}\\n\")\n    f.write(f\"MAPE: {train_mape:.4f}%\\n\")\n    f.write(f\"MAE: {train_mae:.4f}\\n\")\n    f.write(f\"R^2 Score: {train_r2:.4f}\\n\")\n    f.write(f\"Explained Variance Score (EVS): {train_evs:.4f}\\n\")\n    f.write(f\"KGE: {train_kge:.4f}\\n\")\n    \n    f.write(f\"-----\\n\")\n    f.write(f\"Testing Data Evaluation Metrics\\n\")\n    f.write(f\"Timestamp: {readable_time}\\n\\n\")\n    f.write(f\"MSE: {test_mse:.4f}\\n\")\n    f.write(f\"RMSE: {test_rmse:.4f}\\n\")\n    f.write(f\"MAPE: {test_mape:.4f}%\\n\")\n    f.write(f\"MAE: {test_mae:.4f}\\n\")\n    f.write(f\"R^2 Score: {test_r2:.4f}\\n\")\n    f.write(f\"Explained Variance Score (EVS): {test_evs:.4f}\\n\")\n    f.write(f\"KGE: {test_kge:.4f}\\n\")\n\nprint(f\"Metrics saved to: {filename}\")\n\n\n# Plot the data\nplt.figure(figsize=(16,8), dpi=300)\nplt.title('BTC Prediction using Helformer')\nplt.xlabel('Date', fontsize=18)\nplt.ylabel('Close Price', fontsize=18)\n\n# Plot original data\nplt.plot(df['Close'], label='True Values')\n\n# Create a list of NaNs to fill in the missing values in the plot\ntrain_predictions_plot = [np.nan] * len(df)\ntrain_predictions_plot[time_step:training_data_len] = train_predictions\n\n# Add training predictions to the plot\nplt.plot(df.index, train_predictions_plot, color='red', label='Train Predictions')\n\n# Create a list of NaNs to fill in the missing values in the plot\ntest_predictions_plot = [np.nan] * len(df)\ntest_predictions_plot[training_data_len:] = test_predictions\n\n# Add test predictions to the plot\nplt.plot(df.index, test_predictions_plot, color='black', label='Test Predictions')\n\nplt.legend()\n\nplt.savefig (\"/results/Helformer_Predictions.png\")\n\n\n# Assuming observed_values = y_test and predicted_values = test_predictions\nobserved_values = y_test\npredicted_values = test_predictions\n\n# Initialize an array to store returns with transaction cost\nreturns_with_cost = []\n\nfor i in range(1, len(observed_values)):\n    trade_signal = np.sign(predicted_values[i] - observed_values[i-1])\n\n    Rt = np.log(observed_values[i] / observed_values[i-1]) * trade_signal\n\n    # Apply transaction cost\n    Rt_after_cost = Rt - 0.01 * abs(Rt)\n\n    returns_with_cost.append(Rt_after_cost)\n\n# Calculate net value (NV)\nnet_value = np.cumsum(returns_with_cost) + 1\n\n# Calculate total returns\ntotal_return = net_value[-1] - 1\ntotal_return_pcnt = total_return * 100\n\n# Calculate volatility (standard deviation of returns)\nvolatility = np.std(returns_with_cost)\n\n# Calculate max drawdown\ndrawdowns = 1 - net_value / np.maximum.accumulate(net_value)\nmax_drawdown = np.max(drawdowns)\nmax_drawdown_inv = max_drawdown * -1\n\n# Set annual risk-free rate (Rf) to 1%\nrisk_free_rate_annual = 0.01\nrisk_free_rate_daily = risk_free_rate_annual / 365  # Convert to daily risk-free rate\n\n# Calculate Sharpe ratio\nexpected_return = np.mean(returns_with_cost)\nsharpe_ratio = (expected_return - risk_free_rate_daily) / volatility * np.sqrt(365)\n\n# Define file path and name\nfilename = f\"Trading_metrics.txt\"\nfilepath = os.path.join(\"/results\", filename)    #for code ocean\n\n# Write metrics to file\nwith open(filepath, 'w') as f:                 \n    f.write(f\"::     ::\\n\")\n    f.write(f\"Trading Metrics\\n\")\n    f.write(f\"Timestamp: {readable_time}\\n\")\n    f.write(f\"Total Return: {total_return_pcnt:.4f}%\\n\")\n    f.write(f\"Volatility: {volatility:.4f}\\n\")\n    f.write(f\"Max Drawdown: {max_drawdown_inv:.4f}\\n\")\n    f.write(f\"Sharpe Ratio: {sharpe_ratio:.4f}\\n\")\n\nprint(f\"Trading metrics saved to: {filename}\")     # Change filename to filepath for codeocean\n\"\"\"","metadata":{"trusted":true,"execution":{"execution_failed":"2025-07-22T23:11:38.006Z"}},"outputs":[],"execution_count":null}]}