{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport shap\nfrom sklearn.model_selection import KFold\nfrom xgboost import XGBRegressor\nfrom scipy.stats import pearsonr\nimport time\nimport warnings\nimport json\nimport os\nfrom pathlib import Path\nwarnings.filterwarnings('ignore')\n\nclass CFG:\n    train_path = \"/kaggle/input/drw-crypto-market-prediction/train.parquet\"\n    test_path = \"/kaggle/input/drw-crypto-market-prediction/test.parquet\"\n    sample_sub_path = \"/kaggle/input/drw-crypto-market-prediction/sample_submission.csv\"\n    \n    # Progress tracking files\n    progress_file = \"feature_selection_progress.json\"\n    results_file = \"feature_selection_results_incremental.csv\"\n    batch_size = 50  # Number of features to test per run\n\ndef reduce_mem_usage(dataframe, dataset):    \n    print('Reducing memory usage for:', dataset)\n    initial_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    \n    for col in dataframe.columns:\n        col_type = dataframe[col].dtype\n\n        c_min = dataframe[col].min()\n        c_max = dataframe[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                dataframe[col] = dataframe[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                dataframe[col] = dataframe[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                dataframe[col] = dataframe[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                dataframe[col] = dataframe[col].astype(np.int64)\n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                dataframe[col] = dataframe[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                dataframe[col] = dataframe[col].astype(np.float32)\n            else:\n                dataframe[col] = dataframe[col].astype(np.float64)\n\n    final_mem_usage = dataframe.memory_usage().sum() / 1024**2\n    print('--- Memory usage before: {:.2f} MB'.format(initial_mem_usage))\n    print('--- Memory usage after: {:.2f} MB'.format(final_mem_usage))\n    print('--- Decreased memory usage by {:.1f}%\\n'.format(100 * (initial_mem_usage - final_mem_usage) / initial_mem_usage))\n\n    return dataframe\n\ndef create_time_weights(n_samples, decay_factor=0.95):\n    \"\"\"\n    Create exponentially decaying weights based on sample position.\n    More recent samples (higher indices) get higher weights.\n    decay_factor controls the rate of decay (0.95 = 5% decay per time unit)\n    \"\"\"\n    positions = np.arange(n_samples)\n    # Normalize positions to [0, 1] range\n    normalized_positions = positions / (n_samples - 1)\n    # Apply exponential weighting\n    weights = decay_factor ** (1 - normalized_positions)\n    # Normalize weights to sum to n_samples (maintains scale)\n    weights = weights * n_samples / weights.sum()\n    return weights\n\ndef evaluate_feature_set(train_data, features, sample_weights, xgb_params, n_folds=5):\n    \"\"\"\n    Evaluate a feature set using cross-validation and return the average Pearson correlation\n    \"\"\"\n    kf = KFold(n_splits=n_folds, shuffle=True, random_state=42)\n    oof_preds = np.zeros(len(train_data))\n    \n    for fold_idx, (train_idx, valid_idx) in enumerate(kf.split(train_data)):\n        X_train = train_data.iloc[train_idx][features]\n        y_train = train_data.iloc[train_idx][\"label\"]\n        X_valid = train_data.iloc[valid_idx][features]\n        y_valid = train_data.iloc[valid_idx][\"label\"]\n        \n        # Extract sample weights for this fold's training data\n        train_weights = sample_weights[train_idx]\n        \n        model = XGBRegressor(**xgb_params)\n        model.fit(\n            X_train, y_train,\n            sample_weight=train_weights,\n            eval_set=[(X_valid, y_valid)],\n            early_stopping_rounds=25,\n            verbose=0\n        )\n        \n        oof_preds[valid_idx] = model.predict(X_valid)\n    \n    pearson_score = pearsonr(train_data[\"label\"], oof_preds)[0]\n    return pearson_score, oof_preds\n\ndef load_progress():\n    \"\"\"Load previous progress if it exists\"\"\"\n    if os.path.exists(CFG.progress_file):\n        with open(CFG.progress_file, 'r') as f:\n            return json.load(f)\n    return None\n\ndef save_progress(progress_data):\n    \"\"\"Save current progress\"\"\"\n    with open(CFG.progress_file, 'w') as f:\n        json.dump(progress_data, f, indent=2)\n\ndef load_previous_results():\n    \"\"\"Load previous results if they exist\"\"\"\n    if os.path.exists(CFG.results_file):\n        return pd.read_csv(CFG.results_file)\n    return None\n\n# Load data\nprint(\"Loading data...\")\ntrain_full = pd.read_parquet(CFG.train_path).reset_index(drop=True)\ntest_full = pd.read_parquet(CFG.test_path).reset_index(drop=True)\nsample = pd.read_csv(CFG.sample_sub_path)\n\n# Define initial selected features\nselected_features = [\n    \"X863\", \"X856\", \"X344\", \"X598\", \"X862\", \"X385\", \"X852\", \"X603\", \"X860\", \"X674\",\n    \"X415\", \"X345\", \"X137\", \"X855\", \"X174\", \"X302\", \"X178\", \"X532\", \"X168\", \"X612\",\n    \"bid_qty\", \"ask_qty\", \"buy_qty\", \"sell_qty\", \"volume\"\n]\n\n# Identify all available features\nall_features = [col for col in train_full.columns if col not in [\"timestamp\", \"label\"]]\nprint(f\"Total available features: {len(all_features)}\")\n\n# Identify features not in current selection\nadditional_features = [f for f in all_features if f not in selected_features]\nprint(f\"Total features to test: {len(additional_features)}\")\n\n# Load previous progress if exists\nprogress = load_progress()\nif progress:\n    print(\"\\n\" + \"=\"*60)\n    print(\"RESUMING FROM PREVIOUS PROGRESS\")\n    print(\"=\"*60)\n    tested_features = progress.get('tested_features', [])\n    baseline_score = progress.get('baseline_score', None)\n    best_score = progress.get('best_score', baseline_score)\n    best_feature = progress.get('best_feature', None)\n    best_features_list = progress.get('best_features_list', selected_features.copy())\n    \n    # Remove already tested features\n    additional_features = [f for f in additional_features if f not in tested_features]\n    print(f\"Features already tested: {len(tested_features)}\")\n    print(f\"Remaining features to test: {len(additional_features)}\")\n    print(f\"Current best score: {best_score:.6f}\")\n    if best_feature:\n        print(f\"Current best additional feature: {best_feature}\")\nelse:\n    print(\"\\n\" + \"=\"*60)\n    print(\"STARTING FRESH RUN\")\n    print(\"=\"*60)\n    tested_features = []\n    baseline_score = None\n    best_score = None\n    best_feature = None\n    best_features_list = selected_features.copy()\n\n# Prepare data with all features for memory efficiency\ntrain = train_full[all_features + [\"label\"]]\ntest = test_full[all_features]\n\n# Reduce memory usage\ntrain = reduce_mem_usage(train, \"train\")\ntest = reduce_mem_usage(test, \"test\")\n\n# Generate sample weights\nsample_weights = create_time_weights(len(train), decay_factor=0.95)\n\n# XGBoost parameters\nxgb_params = {\n    \"tree_method\": \"gpu_hist\",\n    \"colsample_bylevel\": 0.4778015829774066,\n    \"colsample_bynode\": 0.362764358742407,\n    \"colsample_bytree\": 0.7107423488010493,\n    \"gamma\": 1.7094857725240398,\n    \"learning_rate\": 0.02213323588455387,\n    \"max_depth\": 20,\n    \"max_leaves\": 12,\n    \"min_child_weight\": 16,\n    \"n_estimators\": 1667,\n    \"n_jobs\": -1,\n    \"random_state\": 42,\n    \"reg_alpha\": 39.352415706891264,\n    \"reg_lambda\": 75.44843704068275,\n    \"subsample\": 0.06566669853471274,\n    \"verbosity\": 0\n}\n\n# Evaluate baseline model if not already done\nif baseline_score is None:\n    print(\"\\n\" + \"=\"*60)\n    print(\"EVALUATING BASELINE MODEL\")\n    print(\"=\"*60)\n    start_time = time.time()\n    baseline_score, baseline_oof = evaluate_feature_set(train, selected_features, sample_weights, xgb_params)\n    baseline_time = time.time() - start_time\n    print(f\"Baseline Pearson Correlation: {baseline_score:.6f}\")\n    print(f\"Time taken: {baseline_time:.2f} seconds\")\n    \n    # Initialize best score if not set\n    if best_score is None:\n        best_score = baseline_score\n\n# Load previous results\nprevious_results = load_previous_results()\nif previous_results is not None:\n    results = previous_results.to_dict('records')\n    print(f\"\\nLoaded {len(results)} previous results\")\nelse:\n    results = []\n\n# Select batch of features to test\nfeatures_to_test = additional_features[:CFG.batch_size]\nif not features_to_test:\n    print(\"\\n\" + \"=\"*60)\n    print(\"ALL FEATURES HAVE BEEN TESTED!\")\n    print(\"=\"*60)\n    print(f\"Best score achieved: {best_score:.6f}\")\n    if best_feature:\n        print(f\"Best additional feature: {best_feature}\")\n    # Skip to final model training\nelse:\n    # Feature selection for current batch\n    print(\"\\n\" + \"=\"*60)\n    print(f\"TESTING BATCH OF {len(features_to_test)} FEATURES\")\n    print(\"=\"*60)\n    \n    batch_start_time = time.time()\n    \n    for idx, feature in enumerate(features_to_test):\n        print(f\"\\nTesting feature {idx+1}/{len(features_to_test)} (Total tested: {len(tested_features) + idx + 1}): {feature}\")\n        \n        # Create new feature list with additional feature\n        test_features = selected_features + [feature]\n        \n        try:\n            start_time = time.time()\n            score, _ = evaluate_feature_set(train, test_features, sample_weights, xgb_params)\n            elapsed_time = time.time() - start_time\n            \n            # Calculate improvement\n            improvement = score - baseline_score\n            \n            # Store results\n            results.append({\n                'feature': feature,\n                'score': score,\n                'improvement': improvement,\n                'time': elapsed_time\n            })\n            \n            print(f\"  Score: {score:.6f} | Improvement: {improvement:+.6f} | Time: {elapsed_time:.2f}s\")\n            \n            # Update best if improved\n            if score > best_score:\n                best_score = score\n                best_feature = feature\n                best_features_list = test_features.copy()\n                print(f\"  *** NEW BEST FEATURE! ***\")\n                \n        except Exception as e:\n            print(f\"  Error evaluating feature {feature}: {str(e)}\")\n            continue\n        \n        # Add to tested features\n        tested_features.append(feature)\n    \n    batch_elapsed_time = time.time() - batch_start_time\n    print(f\"\\nBatch completed in {batch_elapsed_time:.2f} seconds\")\n    \n    # Save progress\n    progress_data = {\n        'tested_features': tested_features,\n        'baseline_score': baseline_score,\n        'best_score': best_score,\n        'best_feature': best_feature,\n        'best_features_list': best_features_list,\n        'last_update': time.strftime('%Y-%m-%d %H:%M:%S')\n    }\n    save_progress(progress_data)\n    \n    # Save results\n    results_df = pd.DataFrame(results)\n    results_df = results_df.sort_values('score', ascending=False)\n    results_df.to_csv(CFG.results_file, index=False)\n    print(f\"\\nResults saved to {CFG.results_file}\")\n\n# Show summary of all results so far\nif results:\n    results_df = pd.DataFrame(results)\n    results_df = results_df.sort_values('score', ascending=False)\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"CURRENT FEATURE SELECTION SUMMARY\")\n    print(\"=\"*60)\n    print(f\"\\nBaseline Score: {baseline_score:.6f}\")\n    print(f\"Best Score: {best_score:.6f}\")\n    print(f\"Features tested so far: {len(tested_features)}\")\n    print(f\"Features remaining: {len(additional_features) - len(features_to_test)}\")\n    \n    if best_feature:\n        print(f\"\\nBest Additional Feature: {best_feature}\")\n        print(f\"Improvement: {best_score - baseline_score:+.6f}\")\n    \n    # Show top 10 features\n    print(\"\\nTop 10 Features by Score (from all tested features):\")\n    print(results_df.head(10)[['feature', 'score', 'improvement']].to_string(index=False))\n\n# Check if all features have been tested\nif len(additional_features) == 0:\n    print(\"\\n\" + \"=\"*60)\n    print(\"FEATURE SELECTION COMPLETE - TRAINING FINAL MODEL\")\n    print(\"=\"*60)\n    \n    # Train final model with best feature set\n    FOLDS = 5\n    kf = KFold(n_splits=FOLDS, shuffle=True, random_state=42)\n    oof_preds = np.zeros(len(train))\n    test_preds = np.zeros(len(test))\n    \n    for i, (train_idx, valid_idx) in enumerate(kf.split(train)):\n        print(f\"\\n### Fold {i + 1} ###\")\n        \n        X_train = train.iloc[train_idx][best_features_list]\n        y_train = train.iloc[train_idx][\"label\"]\n        X_valid = train.iloc[valid_idx][best_features_list]\n        y_valid = train.iloc[valid_idx][\"label\"]\n        X_test = test[best_features_list]\n        \n        # Extract sample weights for this fold's training data\n        train_weights = sample_weights[train_idx]\n        \n        model = XGBRegressor(**xgb_params)\n        model.fit(\n            X_train, y_train,\n            sample_weight=train_weights,\n            eval_set=[(X_valid, y_valid)],\n            early_stopping_rounds=25,\n            verbose=100\n        )\n        \n        oof_preds[valid_idx] = model.predict(X_valid)\n        test_preds += model.predict(X_test)\n        \n        fold_score = pearsonr(y_valid, oof_preds[valid_idx])[0]\n        print(f\"Fold {i + 1} Pearson Correlation: {fold_score:.6f}\")\n    \n    # Calculate final score\n    final_score = pearsonr(train[\"label\"], oof_preds)[0]\n    print(f\"\\nFinal Out-of-Fold Pearson Correlation: {final_score:.6f}\")\n    \n    # Average test predictions\n    test_preds /= FOLDS\n    \n    # SHAP analysis\n    print(\"\\nGenerating SHAP analysis...\")\n    explainer = shap.TreeExplainer(model, feature_perturbation=\"tree_path_dependent\", model_output=\"raw\")\n    shap_values = explainer.shap_values(X_test)\n    shap.summary_plot(shap_values, X_test)\n    \n    # Save submission\n    sample[\"prediction\"] = test_preds\n    sample.to_csv(\"submission.csv\", index=False)\n    print(\"\\nPredictions saved to submission.csv\")\n    print(sample.head())\n    \n    # Save feature importance summary\n    feature_importance_summary = pd.DataFrame({\n        'baseline_features': selected_features,\n        'baseline_score': [baseline_score] * len(selected_features) + [np.nan] * (len(best_features_list) - len(selected_features))\n    })\n    if best_feature:\n        feature_importance_summary['best_additional_feature'] = [best_feature if i == 0 else '' for i in range(len(best_features_list))]\n        feature_importance_summary['final_score'] = [best_score if i == 0 else '' for i in range(len(best_features_list))]\n        feature_importance_summary['improvement'] = [best_score - baseline_score if i == 0 else '' for i in range(len(best_features_list))]\n    \n    feature_importance_summary.to_csv(\"feature_importance_summary.csv\", index=False)\n    print(\"\\nFeature importance summary saved to feature_importance_summary.csv\")\n    \n    # Clean up progress file since we're done\n    if os.path.exists(CFG.progress_file):\n        os.remove(CFG.progress_file)\n        print(\"\\nProgress file cleaned up\")\nelse:\n    print(\"\\n\" + \"=\"*60)\n    print(\"BATCH COMPLETE - MORE FEATURES REMAIN\")\n    print(\"=\"*60)\n    print(f\"Run the script again to test the next {CFG.batch_size} features\")\n    print(f\"Progress has been saved to {CFG.progress_file}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}