{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Describe and Learn Data\n","metadata":{}},{"cell_type":"markdown","source":"train.parquet\nThe training dataset containing all historical market data along with the corresponding labels.\n\ntimestamp: The timestamp index representing the minute associated with each row.\nbid_qty: The total quantity buyers are willing to purchase at the best (highest) bid price at the given timestamp.\nask_qty: The total quantity sellers are offering to sell at the best (lowest) ask price at the given timestamp.\nbuy_qty: The total trading quantity executed at the best ask price during the given minute.\nsell_qty: The total trading quantity executed at the best bid price during the given minute.\nvolume: The total traded volume during the minute.\nX_{1,...,890}: A set of anonymized market features derived from proprietary data sources.\nlabel: The target variable representing the anonymized market price movement to be predicted.\n","metadata":{}},{"cell_type":"code","source":"df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\ndf","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(5)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.tail(5)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe().T","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.duplicated().sum()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.fillna(df.mean())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.fillna(df.median())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.fillna(method='ffill')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.resample('H').mean()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['X884'].shift(1) ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['bid_qty'].rolling(window=5).mean()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Veri Tipini Küçültme: Özellikle float64 olan tüm sütunları float32'ye dönüştürmek bellek kullanımını yarı yarıya düşürmesi","metadata":{}},{"cell_type":"code","source":"for col in df.columns:\n    if df[col].dtype == 'float64':\n        df[col] = df[col].astype('float32')\nprint(df.info(memory_usage='deep'))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.dropna(inplace=True)\ndf","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['hour'] = df.index.hour\ndf['dayofweek'] = df.index.dayofweek\n# One-hot encoding for categorical time features if necessary","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\ndf_test","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\n\n\n\n\nprint(\"Orijinal DataFrame (ilk 15 satır):\\n\", df.head(15))\nprint(\"\\nOrijinal DataFrame'deki NaN sayısı:\\n\", df.isnull().sum())\n\n# --- Veriyi kronolojik sıraya göre ayırma ---\ntrain_size = int(len(df) * 0.7)\nval_size = int(len(df) * 0.15)\n# test_size'ı kalan olarak hesaplamak daha güvenlidir.\ntest_size = len(df) - train_size - val_size\n\ntrain_df = df.iloc[:train_size].copy()\nval_df = df.iloc[train_size:train_size + val_size].copy()\ntest_df = df.iloc[train_size + val_size:].copy()\n\nprint(f\"\\nEğitim seti boyutu: {len(train_df)}\")\nprint(f\"Doğrulama seti boyutu: {len(val_df)}\")\nprint(f\"Test seti boyutu: {len(test_df)}\")\n\n# --- Özellikleri ve hedef değişkeni ayırın ---\nfeatures = [col for col in df.columns if col != 'label'] # 'label' hariç tüm sütunlar özellik\ntarget = 'label'\n\nX_train, y_train = train_df[features], train_df[target]\nX_val, y_val = val_df[features], val_df[target]\nX_test, y_test = test_df[features], test_df[target]\n\nprint(\"\\n--- NaN Yönetimi Öncesi ---\")\nprint(\"X_train NaN sayısı:\\n\", X_train.isnull().sum().sum())\nprint(\"X_val NaN sayısı:\\n\", X_val.isnull().sum().sum())\nprint(\"X_test NaN sayısı:\\n\", X_test.isnull().sum().sum())\n\n# --- NaN değerleri yönetin (fillna veya dropna) ---\n# Burada dropna kullandık, ancak duruma göre fillna da tercih edilebilir.\n# Özellikle zaman serilerinde 'ffill' (ileri doldurma) veya 'bfill' (geri doldurma) yaygın kullanılır.\n\nprint(\"\\n--- NaN Yönetimi Sonrası ---\")\n\n# Eğitim seti için NaN yönetimi\ninitial_train_len = len(X_train)\nX_train.dropna(inplace=True)\ny_train = y_train[X_train.index] # X_train'deki indekslerle y_train'i senkronize et\nprint(f\"Eğitim setinden düşen satır sayısı: {initial_train_len - len(X_train)}\")\n\n# Doğrulama seti için NaN yönetimi\ninitial_val_len = len(X_val)\nX_val.dropna(inplace=True)\ny_val = y_val[X_val.index] # X_val'deki indekslerle y_val'i senkronize et\nprint(f\"Doğrulama setinden düşen satır sayısı: {initial_val_len - len(X_val)}\")\n\n# Test seti için NaN yönetimi\ninitial_test_len = len(X_test)\nX_test.dropna(inplace=True)\ny_test = y_test[X_test.index] # X_test'deki indekslerle y_test'i senkronize et\nprint(f\"Test setinden düşen satır sayısı: {initial_test_len - len(X_test)}\")\n\nprint(\"\\n--- Son Durum ---\")\nprint(\"X_train.shape:\", X_train.shape, \"y_train.shape:\", y_train.shape)\nprint(\"X_val.shape:\", X_val.shape, \"y_val.shape:\", y_val.shape)\nprint(\"X_test.shape:\", X_test.shape, \"y_test.shape:\", y_test.shape)\n\nprint(\"\\nSon kontrol - NaN kalmadı mı?\")\nprint(\"X_train NaN sayısı:\\n\", X_train.isnull().sum().sum())\nprint(\"X_val NaN sayısı:\\n\", X_val.isnull().sum().sum())\nprint(\"X_test NaN sayısı:\\n\", X_test.isnull().sum().sum())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport lightgbm as lgb\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.preprocessing import MinMaxScaler # If you decide to use it for deep learning later\n\n# --- 1. Load the data (assuming you've fixed the .parquet issue) ---\n# Your actual path would be: '/kaggle/input/drw-crypto-market-prediction/train.parquet'\ntry:\n    df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\n    print(\"Parquet file loaded successfully!\")\nexcept FileNotFoundError:\n    print(\"Error: The file 'train.parquet' was not found. Please check the path.\")\n    # Exit or handle the error appropriately if the file isn't found\n    exit()\nexcept Exception as e:\n    print(f\"An unexpected error occurred during file loading: {e}\")\n    exit()\n\n# --- 2. (Optional but recommended for large datasets) Memory Optimization ---\n# Convert float64 to float32 to reduce memory usage\nfor col in df.columns:\n    if df[col].dtype == 'float64':\n        df[col] = df[col].astype('float32')\nprint(\"\\nDataFrame info after memory optimization:\")\ndf.info(memory_usage='deep')\n\n# --- 3. Feature Engineering (Example - you will expand on this) ---\n# Make sure 'label' is the target. If you need to create a 'label' first, do it here.\n# For example, if you're predicting the next 5-minute price change:\n# df['price'] = df['mid_price'] # Assuming 'mid_price' exists or you derive it\n# df['label'] = df['price'].shift(-5) - df['price'] # Predict 5-min future price change\n# Or for classification:\n# df['label'] = (df['price'].shift(-5) > df['price']).astype(int) # 1 if price goes up, 0 otherwise\n\n# Example: Add a simple lagged feature (this will introduce NaNs at the beginning)\n# Replace 'some_price_col' with an actual price-like column from your data\n# For instance, if you have 'wap' or 'price' columns, use one of them.\n# Let's assume 'wap' (Weighted Average Price) is a relevant price column.\nif 'wap' in df.columns:\n    df['wap_lag_1'] = df['wap'].shift(1)\n    df['wap_ma_10'] = df['wap'].rolling(window=10).mean()\n    # Ensure 'label' column exists before proceeding\n    if 'label' not in df.columns:\n        print(\"\\n'label' column not found. Creating a dummy label for demonstration.\")\n        # Create a dummy 'label' for demonstration if it doesn't exist\n        # In a real scenario, you define your target variable clearly.\n        df['label'] = df['wap'].shift(-1) # Example: predicting next step's WAP\n        # Drop the last row since the label will be NaN\n        df.dropna(subset=['label'], inplace=True)\nelse:\n    print(\"\\n'wap' column not found. Please ensure you have relevant price columns or create your features.\")\n    # If 'wap' doesn't exist, and you need to create features,\n    # you might need to adjust based on your actual column names.\n    # For now, let's assume a 'label' already exists or we create a dummy.\n    if 'label' not in df.columns:\n        # Fallback for demonstration if no 'wap' and no 'label'\n        print(\"Creating a dummy label from first feature.\")\n        df['label'] = df[df.columns[0]].shift(-1)\n        df.dropna(subset=['label'], inplace=True)\n\n\nprint(\"\\nDataFrame head after initial feature engineering and potential dummy label creation:\")\nprint(df.head())\nprint(\"\\nNaNs after feature engineering (before splitting):\")\nprint(df.isnull().sum().sum()) # Total NaNs in the DataFrame\n\n# --- 4. Data Splitting and NaN Management ---\n\n# Veriyi kronolojik sıraya göre ayırma\ntrain_size = int(len(df) * 0.7)\nval_size = int(len(df) * 0.15)\ntest_size = len(df) - train_size - val_size # Safely calculate remaining for test\n\ntrain_df = df.iloc[:train_size].copy()\nval_df = df.iloc[train_size:train_size + val_size].copy()\ntest_df = df.iloc[train_size + val_size:].copy()\n\nprint(f\"\\nEğitim seti boyutu: {len(train_df)}\")\nprint(f\"Doğrulama seti boyutu: {len(val_df)}\")\nprint(f\"Test seti boyutu: {len(test_df)}\")\n\n# Özellikleri ve hedef değişkeni ayırın\nfeatures = [col for col in df.columns if col != 'label']\ntarget = 'label'\n\nX_train, y_train = train_df[features], train_df[target]\nX_val, y_val = val_df[features], val_df[target]\nX_test, y_test = test_df[features], test_df[target]\n\nprint(\"\\n--- NaN Yönetimi Öncesi ---\")\nprint(\"X_train NaN sayısı:\", X_train.isnull().sum().sum())\nprint(\"X_val NaN sayısı:\", X_val.isnull().sum().sum())\nprint(\"X_test NaN sayısı:\", X_test.isnull().sum().sum())\n\n# NaN değerleri yönetin (fillna veya dropna)\n# Önemli: lagged özellikler nedeniyle başta NaN'lar olabilir.\n# `dropna` yerine `ffill` veya `bfill` de kullanmayı düşünebilirsiniz,\n# özellikle çok fazla satır kaybetmek istemiyorsanız.\n# Burada 'dropna' kullanalım:\nprint(\"\\n--- NaN Yönetimi Sonrası ---\")\n\n# Eğitim seti için NaN yönetimi\ninitial_train_len = len(X_train)\nX_train.dropna(inplace=True)\ny_train = y_train[X_train.index] # X_train'deki indekslerle y_train'i senkronize et\nprint(f\"Eğitim setinden düşen satır sayısı: {initial_train_len - len(X_train)}\")\n\n# Doğrulama seti için NaN yönetimi\ninitial_val_len = len(X_val)\nX_val.dropna(inplace=True)\ny_val = y_val[X_val.index] # X_val'deki indekslerle y_val'i senkronize et\nprint(f\"Doğrulama setinden düşen satır sayısı: {initial_val_len - len(X_val)}\")\n\n# Test seti için NaN yönetimi\ninitial_test_len = len(X_test)\nX_test.dropna(inplace=True)\ny_test = y_test[X_test.index] # X_test'deki indekslerle y_test'i senkronize et\nprint(f\"Test setinden düşen satır sayısı: {initial_test_len - len(X_test)}\")\n\nprint(\"\\n--- Son Durum (Shapes) ---\")\nprint(\"X_train.shape:\", X_train.shape, \"y_train.shape:\", y_train.shape)\nprint(\"X_val.shape:\", X_val.shape, \"y_val.shape:\", y_val.shape)\nprint(\"X_test.shape:\", X_test.shape, \"y_test.shape:\", y_test.shape)\n\nprint(\"\\nSon kontrol - NaN kalmadı mı?\")\nprint(\"X_train NaN sayısı:\", X_train.isnull().sum().sum())\nprint(\"X_val NaN sayısı:\", X_val.isnull().sum().sum())\nprint(\"X_test NaN sayısı:\", X_test.isnull().sum().sum())\n\n\n# --- 5. Model Training (LightGBM Regression Example) ---\n# Make sure your 'label' is continuous for regression. If it's 0/1, use Classifier.\n\n# Ensure all columns in X_train, X_val, X_test are numeric\n# (LightGBM generally handles this well, but good to check if you have object columns)\nfor col in X_train.columns:\n    if X_train[col].dtype == 'object':\n        print(f\"Warning: Column '{col}' in X_train is object type. Consider converting to numeric.\")\n\n\nprint(\"\\nStarting LightGBM Model Training...\")\n# Modeli tanımlama\n# Assuming your 'label' column is a continuous value (regression task)\nlgbm_reg = lgb.LGBMRegressor(objective='regression_l1', # Use 'regression_l1' for MAE, 'regression' for MSE\n                             n_estimators=1000,\n                             learning_rate=0.05,\n                             num_leaves=31, # A common default, tune this\n                             random_state=42,\n                             n_jobs=-1) # Use all available CPU cores\n\n# Modeli eğitme\n# eval_metric should match objective for consistency, or use a relevant one for your evaluation.\n# 'mae' is suitable for 'regression_l1' objective.\nlgbm_reg.fit(X_train, y_train,\n             eval_set=[(X_val, y_val)],\n             eval_metric='mae',\n             callbacks=[lgb.early_stopping(100, verbose=True)]) # verbose=True will print progress\n\n# Tahminler yapma\nprint(\"\\nMaking predictions on the test set...\")\npredictions = lgbm_reg.predict(X_test)\n\n# Model performansı\nrmse = mean_squared_error(y_test, predictions, squared=False)\nr2 = r2_score(y_test, predictions)\nprint(f\"Test RMSE: {rmse}\")\nprint(f\"Test R2 Score: {r2}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.metrics import mean_squared_error, r2_score\n\n# Modeli tanımlama\nlgbm_reg = lgb.LGBMRegressor(objective='regression_l1', # MAE (Mean Absolute Error) için\n                             n_estimators=1000,\n                             learning_rate=0.05,\n                             num_leaves=31,\n                             random_state=42,\n                             n_jobs=-1) # Tüm çekirdekleri kullan\n\n# Modeli eğitme\nlgbm_reg.fit(X_train, y_train,\n             eval_set=[(X_val, y_val)],\n             eval_metric='mae', # Değerlendirme metriği\n             callbacks=[lgb.early_stopping(100, verbose=False)]) # Erken durdurma\n\n# Tahminler yapma\npredictions = lgbm_reg.predict(X_test)\n\n# Model performansı\nrmse = mean_squared_error(y_test, predictions, squared=False)\nr2 = r2_score(y_test, predictions)\nprint(f\"Test RMSE: {rmse}\")\nprint(f\"Test R2 Score: {r2}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.metrics import accuracy_score, classification_report, roc_auc_score\n\n# Modeli tanımlama (ikili sınıflandırma için)\nlgbm_clf = lgb.LGBMClassifier(objective='binary', # İkili sınıflandırma\n                              n_estimators=1000,\n                              learning_rate=0.05,\n                              num_leaves=31,\n                              random_state=42,\n                              n_jobs=-1)\n\n# Modeli eğitme\nlgbm_clf.fit(X_train, y_train,\n             eval_set=[(X_val, y_val)],\n             eval_metric='auc', # AUC metriği\n             callbacks=[lgb.early_stopping(100, verbose=False)])\n\n# Tahminler yapma (sınıf etiketleri ve olasılıklar)\npredictions_classes = lgbm_clf.predict(X_test)\npredictions_proba = lgbm_clf.predict_proba(X_test)[:, 1] # Pozitif sınıf olasılıkları\n\n# Model performansı\naccuracy = accuracy_score(y_test, predictions_classes)\nroc_auc = roc_auc_score(y_test, predictions_proba)\nprint(f\"Test Accuracy: {accuracy}\")\nprint(f\"Test ROC AUC: {roc_auc}\")\nprint(f\"Classification Report:\\n{classification_report(y_test, predictions_classes)}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import LSTM, Dense, Dropout\nfrom sklearn.preprocessing import MinMaxScaler\nimport numpy as np\n\n# Veriyi ölçeklendirme (derin öğrenme için önemli)\nscaler = MinMaxScaler(feature_range=(0, 1))\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)\nX_test_scaled = scaler.transform(X_test)\n\n# LSTM için veriyi yeniden şekillendirme: (örnek sayısı, zaman adımı, özellik sayısı)\n# Örneğin, her bir tahmin için son 60 dakikanın verisini kullanmak\ntime_steps = 60 # Kaç geçmiş zaman adımını modele besleyeceğiniz\nX_train_lstm = []\ny_train_lstm = []\nfor i in range(time_steps, len(X_train_scaled)):\n    X_train_lstm.append(X_train_scaled[i-time_steps:i, :])\n    y_train_lstm.append(y_train.iloc[i]) # Hedef değişkenin i. indeksi\nX_train_lstm, y_train_lstm = np.array(X_train_lstm), np.array(y_train_lstm)\n\n# Doğrulama ve test kümeleri için de benzer işlemi yapın\n\n# LSTM modelini oluşturma\nmodel = Sequential()\nmodel.add(LSTM(units=50, return_sequences=True, input_shape=(X_train_lstm.shape[1], X_train_lstm.shape[2])))\nmodel.add(Dropout(0.2))\nmodel.add(LSTM(units=50))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(units=1)) # Regresyon için 1 çıkış, sınıflandırma için uygun aktivasyon ve çıkış katmanı\n\nmodel.compile(optimizer='adam', loss='mean_squared_error') # Regresyon için\n# Sınıflandırma için: model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Modeli eğitme\nhistory = model.fit(X_train_lstm, y_train_lstm, epochs=50, batch_size=32, validation_data=(X_val_lstm, y_val_lstm), verbose=1)\n\n# Tahmin yapma\n# predictions = model.predict(X_test_lstm)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}