{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-06-08T17:54:03.764642Z","iopub.execute_input":"2025-06-08T17:54:03.764945Z","iopub.status.idle":"2025-06-08T17:54:06.322238Z","shell.execute_reply.started":"2025-06-08T17:54:03.764908Z","shell.execute_reply":"2025-06-08T17:54:06.320950Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import cudf\n# import cupy\n# from xgboost import XGBRegressor\n# import lightgbm as lgb\n# import catboost as cat\n# from catboost import CatBoostClassifier\n# from catboost import CatBoostRegressor\nfrom sklearn import linear_model\nfrom sklearn.decomposition import PCA\n\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.linear_model import RidgeCV\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom scipy.stats import pearsonr\nimport pickle\nimport gc\n\nimport warnings\n\nwarnings.filterwarnings('ignore')\n\n# from tqdm import tqdm\n# cudf.__version__","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T18:34:14.942196Z","iopub.execute_input":"2025-06-08T18:34:14.942634Z","iopub.status.idle":"2025-06-08T18:34:14.949357Z","shell.execute_reply.started":"2025-06-08T18:34:14.942607Z","shell.execute_reply":"2025-06-08T18:34:14.947959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# one of most copied functions on Kaggle\ndef reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T17:55:33.719132Z","iopub.execute_input":"2025-06-08T17:55:33.719536Z","iopub.status.idle":"2025-06-08T17:55:33.731529Z","shell.execute_reply.started":"2025-06-08T17:55:33.719515Z","shell.execute_reply":"2025-06-08T17:55:33.730233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_raw = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ntest_raw = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T17:55:40.321238Z","iopub.execute_input":"2025-06-08T17:55:40.321573Z","iopub.status.idle":"2025-06-08T17:56:55.051388Z","shell.execute_reply.started":"2025-06-08T17:55:40.321547Z","shell.execute_reply":"2025-06-08T17:56:55.049584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = reduce_mem_usage(train_raw)\ntest = reduce_mem_usage(test_raw)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T17:57:30.882654Z","iopub.execute_input":"2025-06-08T17:57:30.882973Z","iopub.status.idle":"2025-06-08T17:57:47.578199Z","shell.execute_reply.started":"2025-06-08T17:57:30.882948Z","shell.execute_reply":"2025-06-08T17:57:47.577235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T17:57:47.579475Z","iopub.execute_input":"2025-06-08T17:57:47.579757Z","iopub.status.idle":"2025-06-08T17:57:47.642315Z","shell.execute_reply.started":"2025-06-08T17:57:47.579730Z","shell.execute_reply":"2025-06-08T17:57:47.641220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.tail(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:59:22.094243Z","iopub.execute_input":"2025-05-25T17:59:22.094511Z","iopub.status.idle":"2025-05-25T17:59:22.132486Z","shell.execute_reply.started":"2025-05-25T17:59:22.094489Z","shell.execute_reply":"2025-05-25T17:59:22.131455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T17:59:22.133396Z","iopub.execute_input":"2025-05-25T17:59:22.133641Z","iopub.status.idle":"2025-05-25T17:59:22.172749Z","shell.execute_reply.started":"2025-05-25T17:59:22.133623Z","shell.execute_reply":"2025-05-25T17:59:22.171739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del train_raw, test_raw\ngc.collect()\n\nDEBUG = False\nif DEBUG:\n    train = train.iloc[:10000]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T17:57:47.643625Z","iopub.execute_input":"2025-06-08T17:57:47.643995Z","iopub.status.idle":"2025-06-08T17:57:47.752381Z","shell.execute_reply.started":"2025-06-08T17:57:47.643965Z","shell.execute_reply":"2025-06-08T17:57:47.751122Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Features","metadata":{}},{"cell_type":"code","source":"# copied from https://www.kaggle.com/code/suthcong/drw-cpm-lightgbm\n# Drop columns have exactly 1 value\nif False:\n    NUNIQUE1=[c for c in train.columns if train[c].nunique()==1]\n    train.drop(NUNIQUE1,axis=1,inplace=True)\n    test.drop(NUNIQUE1+['label'],axis=1,inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T12:07:18.212003Z","iopub.execute_input":"2025-06-08T12:07:18.212267Z","iopub.status.idle":"2025-06-08T12:07:18.216868Z","shell.execute_reply.started":"2025-06-08T12:07:18.212250Z","shell.execute_reply":"2025-06-08T12:07:18.215981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"NUNIQUE1 = ['X697', 'X698', 'X699', 'X700', 'X701', 'X702', 'X703', 'X704', 'X705', 'X706', 'X707', 'X708', 'X709', 'X710', 'X711', 'X712', 'X713', 'X714', 'X715', 'X716', 'X717', 'X864', 'X867', 'X869', 'X870', 'X871', 'X872']\ntrain.drop(NUNIQUE1,axis=1,inplace=True)\ntest.drop(NUNIQUE1+['label'],axis=1,inplace=True)\nprint(NUNIQUE1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T17:57:47.753572Z","iopub.execute_input":"2025-06-08T17:57:47.753973Z","iopub.status.idle":"2025-06-08T17:57:52.102998Z","shell.execute_reply.started":"2025-06-08T17:57:47.753932Z","shell.execute_reply":"2025-06-08T17:57:52.102018Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature selection using RidgeCV","metadata":{}},{"cell_type":"code","source":"# Training data\nfeatures = [col for col in train if col.startswith('X')]\ntarget = ['label']\nX = train[features]\ny = train['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:21:11.740261Z","iopub.execute_input":"2025-06-08T19:21:11.740713Z","iopub.status.idle":"2025-06-08T19:21:14.586216Z","shell.execute_reply.started":"2025-06-08T19:21:11.740676Z","shell.execute_reply":"2025-06-08T19:21:14.585262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(X.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T17:58:14.520581Z","iopub.execute_input":"2025-06-08T17:58:14.521418Z","iopub.status.idle":"2025-06-08T17:58:14.526698Z","shell.execute_reply.started":"2025-06-08T17:58:14.521387Z","shell.execute_reply":"2025-06-08T17:58:14.525493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"alphas=np.logspace(-6, 6, num=3)\nprint(alphas)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:21:42.604669Z","iopub.execute_input":"2025-06-08T19:21:42.605068Z","iopub.status.idle":"2025-06-08T19:21:42.613054Z","shell.execute_reply.started":"2025-06-08T19:21:42.605041Z","shell.execute_reply":"2025-06-08T19:21:42.611551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if False:\n    ridge = RidgeCV(alphas=np.logspace(-6, 6, num=3)).fit(X, y)\n    importance = np.abs(ridge.coef_)\n    \n    # feature_names = np.array(X.feature_names)\n    df_imp = pd.DataFrame()\n    df_imp['names'] = X.columns\n    df_imp['importance'] = importance\n    df_imp = df_imp.sort_values(by='importance', ascending=False)\n    #\n    imp_features = df_imp['names'].head(300).to_list()\n    print(imp_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:21:48.121789Z","iopub.execute_input":"2025-06-08T19:21:48.122213Z","iopub.status.idle":"2025-06-08T19:23:41.573217Z","shell.execute_reply.started":"2025-06-08T19:21:48.122185Z","shell.execute_reply":"2025-06-08T19:23:41.571875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imp_features = ['X40', 'X289', 'X42', 'X292', 'X296', 'X53', 'X34', 'X287', 'X45', 'X297', 'X55', 'X301', 'X36', 'X48', 'X54', 'X283', 'X295', 'X286', 'X797', 'X47', 'X303', 'X291', 'X694', 'X52', 'X691', 'X688', 'X685', 'X288', 'X46', 'X801', 'X793', 'X798', 'X679', 'X673', 'X741', 'X794', 'X837', 'X833', 'X724', 'X682', 'X742', 'X723', 'X725', 'X745', 'X124', 'X829', 'X825', 'X281', 'X802', 'X201', 'X56', 'X33', 'X729', 'X676', 'X129', 'X451', 'X285', 'X166', 'X282', 'X727', 'X722', 'X35', 'X450', 'X746', 'X49', 'X254', 'X821', 'X796', 'X130', 'X458', 'X15', 'X726', 'X749', 'X215', 'X452', 'X822', 'X826', 'X200', 'X277', 'X172', 'X806', 'X832', 'X14', 'X738', 'X240', 'X486', 'X728', 'X750', 'X247', 'X255', 'X889', 'X194', 'X737', 'X810', 'X720', 'X670', 'X41', 'X193', 'X753', 'X284', 'X789', 'X830', 'X493', 'X449', 'X457', 'X888', 'X800', 'X455', 'X239', 'X445', 'X448', 'X136', 'X223', 'X652', 'X119', 'X214', 'X444', 'X643', 'X721', 'X790', 'X278', 'X275', 'X637', 'X887', 'X805', 'X22', 'X246', 'X838', 'X731', 'X409', 'X485', 'X222', 'X420', 'X841', 'X253', 'X447', 'X459', 'X238', 'X135', 'X781', 'X276', 'X886', 'X421', 'X640', 'X836', 'X204', 'X123', 'X777', 'X39', 'X730', 'X163', 'X834', 'X469', 'X490', 'X380', 'X160', 'X754', 'X118', 'X16', 'X817', 'X51', 'X50', 'X492', 'X131', 'X483', 'X202', 'X175', 'X812', 'X178', 'X43', 'X818', 'X23', 'X262', 'X646', 'X658', 'X808', 'X470', 'X335', 'X70', 'X293', 'X403', 'X442', 'X245', 'X664', 'X261', 'X154', 'X28', 'X484', 'X156', 'X415', 'X471', 'X386', 'X824', 'X491', 'X7', 'X270', 'X113', 'X441', 'X463', 'X125', 'X414', 'X465', 'X377', 'X443', 'X225', 'X331', 'X828', 'X112', 'X6', 'X157', 'X171', 'X44', 'X224', 'X487', 'X744', 'X81', 'X456', 'X332', 'X21', 'X300', 'X298', 'X323', 'X402', 'X773', 'X27', 'X329', 'X338', 'X237', 'X792', 'X208', 'X454', 'X199', 'X169', 'X87', 'X494', 'X82', 'X5', 'X76', 'X813', 'X427', 'X740', 'X466', 'X13', 'X197', 'X260', 'X294', 'X757', 'X302', 'X29', 'X267', 'X748', 'X785', 'X221', 'X192', 'X464', 'X774', 'X655', 'X780', 'X30', 'X162', 'X173', 'X677', 'X244', 'X842', 'X649', 'X325', 'X133', 'X667', 'X336', 'X371', 'X446', 'X205', 'X341', 'X778', 'X752', 'X814', 'X168', 'X376', 'X840', 'X72', 'X695', 'X232', 'X603', 'X661', 'X882', 'X93', 'X334', 'X328', 'X203', 'X342', 'X177', 'X440', 'X765', 'X326', 'X383', 'X181', 'X408', 'X139', 'X236', 'X75']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:24:37.511014Z","iopub.execute_input":"2025-06-08T19:24:37.511398Z","iopub.status.idle":"2025-06-08T19:24:37.524107Z","shell.execute_reply.started":"2025-06-08T19:24:37.511374Z","shell.execute_reply":"2025-06-08T19:24:37.523046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# imp_features = imp_features[0:49]\n# imp_features = imp_features[0:99]\n# imp_features = imp_features[0:149]\n# imp_features = imp_features[0:49]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:24:45.284127Z","iopub.execute_input":"2025-06-08T19:24:45.284429Z","iopub.status.idle":"2025-06-08T19:24:45.289542Z","shell.execute_reply.started":"2025-06-08T19:24:45.284410Z","shell.execute_reply":"2025-06-08T19:24:45.288397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(imp_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:24:49.428329Z","iopub.execute_input":"2025-06-08T19:24:49.428661Z","iopub.status.idle":"2025-06-08T19:24:49.434462Z","shell.execute_reply.started":"2025-06-08T19:24:49.428637Z","shell.execute_reply":"2025-06-08T19:24:49.433113Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# NN","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Training data\nvol_features = ['bid_qty','ask_qty','buy_qty','sell_qty','volume']\nfeatures = train.columns[train.columns!='label'] # vol_features + imp_features\ntarget = ['label']\nX_train = train[features]\n\nX_test = test[features]\ny = train['label']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:25:02.673894Z","iopub.execute_input":"2025-06-08T19:25:02.674238Z","iopub.status.idle":"2025-06-08T19:25:03.682402Z","shell.execute_reply.started":"2025-06-08T19:25:02.674214Z","shell.execute_reply":"2025-06-08T19:25:03.681416Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Standardize the data\nscaler = StandardScaler() # RobustScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:25:11.671114Z","iopub.execute_input":"2025-06-08T19:25:11.671433Z","iopub.status.idle":"2025-06-08T19:25:23.004873Z","shell.execute_reply.started":"2025-06-08T19:25:11.671411Z","shell.execute_reply":"2025-06-08T19:25:23.004016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# PCA\nK = 25 # 50 \nif True:\n    pca = PCA(n_components=K)\n    X_train_scaled = pca.fit_transform(X_train_scaled)\n    X_test_scaled = pca.transform(X_test_scaled)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a preprocessor for numerical and categorical features\n# preprocessor = ColumnTransformer(\n#     transformers=[\n#         ('num', StandardScaler(), features), # Scale numerical features\n#         # ('cat', OneHotEncoder(handle_unknown='ignore'), categorical_cols) # One-hot encode categorical features\n#    ])\n\n# Fit and transform the training data\n# X_train_processed = preprocessor.fit_transform(X_train)\n# Transform the test data using the fitted preprocessor\n# X_valid_processed = preprocessor.transform(X_valid)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 2. Train-Valid Split ---\nX_train, X_valid, y_train, y_valid = train_test_split(X_train_scaled, y, test_size=0.2, shuffle = True, random_state=42)\n\n# Get the number of input features for the NN\ninput_shape = X_train.shape[1]\nprint(f\"\\nNumber of input features after preprocessing: {input_shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:25:27.463528Z","iopub.execute_input":"2025-06-08T19:25:27.463904Z","iopub.status.idle":"2025-06-08T19:25:34.448683Z","shell.execute_reply.started":"2025-06-08T19:25:27.463877Z","shell.execute_reply":"2025-06-08T19:25:34.447560Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 4. Build the Neural Network Model ---\nDROP = 0.5\ndef build_nn_model(input_shape):\n    model = keras.Sequential([\n        layers.Input(shape=(input_shape,)), # Input layer, matches the number of preprocessed features\n        # layers.Dense(256, activation='relu'), # First hidden layer with ReLU activation\n        # layers.Dropout(DROP), # Dropout for regularization\n        layers.Dense(128, activation='relu'), # First hidden layer with ReLU activation\n        layers.Dropout(DROP), # Dropout for regularization\n        layers.Dense(64, activation='relu'),  # Second hidden layer\n        layers.Dropout(DROP),\n        layers.Dense(32, activation='relu'),  # Third hidden layer\n        layers.Dense(1) # Output layer with a single neuron for regression (no activation for linear output)\n    ])\n\n    # Compile the model # Adam optimizer is a good default \n    # 'mse' (Mean Squared Error) is a common loss function for regression\n    model.compile(optimizer='adam', loss='mse', metrics=['mae']) # mae for Mean Absolute Error as a metric\n    return model\n\nnn_model = build_nn_model(input_shape)\nnn_model.summary()\n\n# --- 5. Train the Neural Network ---\n# Use EarlyStopping to prevent overfitting and save the best model\nearly_stopping = keras.callbacks.EarlyStopping(\n    monitor='val_loss', # Monitor validation loss\n    patience=10,        # Number of epochs with no improvement after which training will be stopped\n    restore_best_weights=True # Restore model weights from the epoch with the best value of the monitored quantity\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:25:37.255455Z","iopub.execute_input":"2025-06-08T19:25:37.255796Z","iopub.status.idle":"2025-06-08T19:25:37.325829Z","shell.execute_reply.started":"2025-06-08T19:25:37.255770Z","shell.execute_reply":"2025-06-08T19:25:37.324865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = nn_model.fit(\n    X_train, y_train,\n    epochs=100, # Max epochs, but EarlyStopping will likely stop it sooner\n    batch_size=512, # Number of samples per gradient update\n    validation_split=0.1, # Use a portion of the training data for validation during training\n    callbacks=[early_stopping],\n    verbose=1 # Show training progress\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:25:43.819983Z","iopub.execute_input":"2025-06-08T19:25:43.820328Z","iopub.status.idle":"2025-06-08T19:26:54.219621Z","shell.execute_reply.started":"2025-06-08T19:25:43.820302Z","shell.execute_reply":"2025-06-08T19:26:54.218407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 6. Evaluate the Model on valid Data ---\nprint(\"\\nEvaluating the model on the valid set...\")\nloss, mae = nn_model.evaluate(X_valid, y_valid, verbose=0)\nprint(f\"Test Loss (MSE): {loss:.4f}\")\nprint(f\"Test MAE: {mae:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:26:54.222132Z","iopub.execute_input":"2025-06-08T19:26:54.222432Z","iopub.status.idle":"2025-06-08T19:27:01.420065Z","shell.execute_reply.started":"2025-06-08T19:26:54.222409Z","shell.execute_reply":"2025-06-08T19:27:01.418942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = nn_model.predict(X_valid).flatten() # Flatten output for metrics\n\nrmse = np.sqrt(mean_squared_error(y_valid, y_pred))\ncorr = np.corrcoef(np.array(y_valid).flatten(), np.array(y_pred).flatten())[0, 1]\nprint(f\"Pearson correlation: {corr}\") \nprint(f\"Test RMSE: {rmse:.4f}\")\n\n# features 200, (256,64,32) Pearson correlation: 0.057588406374714775\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:27:08.028131Z","iopub.execute_input":"2025-06-08T19:27:08.028538Z","iopub.status.idle":"2025-06-08T19:27:14.393738Z","shell.execute_reply.started":"2025-06-08T19:27:08.028515Z","shell.execute_reply":"2025-06-08T19:27:14.392616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- 7. Visualize Training History (Optional) ---\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nplt.plot(history.history['loss'], label='Train Loss')\nplt.plot(history.history['val_loss'], label='Validation Loss')\nplt.title('Model Loss Over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('Loss (MSE)')\nplt.legend()\nplt.grid(True)\n\nplt.subplot(1, 2, 2)\nplt.plot(history.history['mae'], label='Train MAE')\nplt.plot(history.history['val_mae'], label='Validation MAE')\nplt.title('Model MAE Over Epochs')\nplt.xlabel('Epoch')\nplt.ylabel('MAE')\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:27:47.435326Z","iopub.execute_input":"2025-06-08T19:27:47.435713Z","iopub.status.idle":"2025-06-08T19:27:47.928151Z","shell.execute_reply.started":"2025-06-08T19:27:47.435683Z","shell.execute_reply":"2025-06-08T19:27:47.926900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"preds = nn_model.predict(X_test_scaled).flatten() # Flatten output for metrics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:35:28.783457Z","iopub.execute_input":"2025-06-08T19:35:28.783941Z","iopub.status.idle":"2025-06-08T19:36:05.965268Z","shell.execute_reply.started":"2025-06-08T19:35:28.783911Z","shell.execute_reply":"2025-06-08T19:36:05.964081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\")\nsubmission[\"prediction\"] = preds\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-08T19:36:37.068480Z","iopub.execute_input":"2025-06-08T19:36:37.068882Z","iopub.status.idle":"2025-06-08T19:36:38.391555Z","shell.execute_reply.started":"2025-06-08T19:36:37.068850Z","shell.execute_reply":"2025-06-08T19:36:38.390534Z"}},"outputs":[],"execution_count":null}]}