{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":12993472,"sourceType":"competition"},{"sourceId":249869065,"sourceType":"kernelVersion"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-12T07:26:06.170465Z","iopub.execute_input":"2025-07-12T07:26:06.170651Z","iopub.status.idle":"2025-07-12T07:26:08.185972Z","shell.execute_reply.started":"2025-07-12T07:26:06.170635Z","shell.execute_reply":"2025-07-12T07:26:08.185253Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import Ridge\nfrom sklearn.ensemble import VotingRegressor, RandomForestRegressor\nfrom sklearn.model_selection import TimeSeriesSplit, cross_val_score\nfrom sklearn.metrics import mean_squared_error\nimport xgboost as xgb\nimport lightgbm as lgb\nimport os\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\ndef optimize_memory(df, verbose=True):\n    \"\"\"\n    Optimize memory usage by downcasting numeric types where possible.\n    \"\"\"\n    if verbose:\n        start_mem = df.memory_usage().sum() / 1024**2\n        print(f'Memory usage before optimization: {start_mem:.2f} MB')\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != 'object':\n            c_min = df[col].min()\n            c_max = df[col].max()\n            \n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                # For float types, we'll use float32 instead of float64\n                # This is usually sufficient for ML models\n                if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n    \n    if verbose:\n        end_mem = df.memory_usage().sum() / 1024**2\n        print(f'Memory usage after optimization: {end_mem:.2f} MB')\n        print(f'Decreased by {100 * (start_mem - end_mem) / start_mem:.1f}%')\n    \n    return df\n\n\ndef create_advanced_features(df, top_features):\n    \"\"\"\n    Create advanced engineered features including interactions, rolling stats, and technical indicators.\n    \"\"\"\n    print(\"Creating advanced features...\")\n    \n    # Rolling statistics for top features\n    windows = [5, 10, 20, 30]\n    for feature in top_features[:10]:  # Only for top 10 to manage memory\n        if feature in df.columns:\n            for window in windows:\n                # Rolling mean\n                df[f'{feature}_ma_{window}'] = df[feature].rolling(window, min_periods=1).mean().astype(np.float32)\n                # Rolling std (volatility)\n                df[f'{feature}_vol_{window}'] = df[feature].rolling(window, min_periods=1).std().fillna(0).astype(np.float32)\n                # Rolling min/max\n                df[f'{feature}_min_{window}'] = df[feature].rolling(window, min_periods=1).min().astype(np.float32)\n                df[f'{feature}_max_{window}'] = df[feature].rolling(window, min_periods=1).max().astype(np.float32)\n    \n    # Feature interactions for top features\n    print(\"Creating feature interactions...\")\n    top_5_features = [f for f in top_features[:5] if f in df.columns]\n    \n    for i in range(len(top_5_features)):\n        for j in range(i+1, len(top_5_features)):\n            feat1, feat2 = top_5_features[i], top_5_features[j]\n            \n            # Ratio features\n            df[f'{feat1}_{feat2}_ratio'] = (df[feat1] / (df[feat2].abs() + 1e-8)).astype(np.float32)\n            # Difference features\n            df[f'{feat1}_{feat2}_diff'] = (df[feat1] - df[feat2]).astype(np.float32)\n            # Product features\n            df[f'{feat1}_{feat2}_prod'] = (df[feat1] * df[feat2]).astype(np.float32)\n    \n    # Technical indicators\n    print(\"Creating technical indicators...\")\n    for feature in top_features[:5]:\n        if feature in df.columns:\n            # RSI-like indicator\n            delta = df[feature].diff()\n            gain = delta.where(delta > 0, 0).rolling(14, min_periods=1).mean()\n            loss = (-delta.where(delta < 0, 0)).rolling(14, min_periods=1).mean()\n            rs = gain / (loss + 1e-8)\n            df[f'{feature}_rsi'] = (100 - (100 / (1 + rs))).astype(np.float32)\n            \n            # Momentum\n            df[f'{feature}_momentum_5'] = (df[feature] - df[feature].shift(5)).fillna(0).astype(np.float32)\n            df[f'{feature}_momentum_20'] = (df[feature] - df[feature].shift(20)).fillna(0).astype(np.float32)\n    \n    return df\n\ndef create_ensemble_models():\n    \"\"\"\n    Create ensemble of different model types.\n    \"\"\"\n    models = {\n        'ridge': Ridge(alpha=1.0, copy_X=False),\n        'ridge_l2': Ridge(alpha=5.0, copy_X=False),\n        'xgb': xgb.XGBRegressor(\n            n_estimators=100,\n            max_depth=6,\n            learning_rate=0.1,\n            subsample=0.8,\n            colsample_bytree=0.8,\n            random_state=42,\n            n_jobs=-1,\n            tree_method='hist'\n        )\n    }\n    \n    return models\n\ndef validate_models_time_series(models, X, y, n_splits=3):\n    \"\"\"\n    Validate models using time series split to avoid data leakage.\n    \"\"\"\n    print(\"Performing time series cross-validation...\")\n    \n    # Use time series split to respect temporal order\n    tscv = TimeSeriesSplit(n_splits=n_splits)\n    \n    results = {}\n    \n    for name, model in models.items():\n        print(f\"Validating {name}...\")\n        scores = []\n        \n        for fold, (train_idx, val_idx) in enumerate(tscv.split(X)):\n            X_train_fold = X.iloc[train_idx]\n            X_val_fold = X.iloc[val_idx]\n            y_train_fold = y[train_idx]\n            y_val_fold = y[val_idx]\n            \n            # Fit model\n            model.fit(X_train_fold, y_train_fold)\n            \n            # Predict\n            y_pred = model.predict(X_val_fold)\n            \n            # Calculate correlation (since that's the evaluation metric)\n            correlation = np.corrcoef(y_val_fold, y_pred)[0, 1]\n            scores.append(correlation)\n            \n            print(f\"  Fold {fold+1}: Correlation = {correlation:.4f}\")\n        \n        avg_score = np.mean(scores)\n        std_score = np.std(scores)\n        results[name] = {'mean': avg_score, 'std': std_score, 'scores': scores}\n        \n        print(f\"  {name} - Mean Correlation: {avg_score:.4f} ± {std_score:.4f}\")\n    \n    return results\n\n\ndef preprocess_data_chunked(raw_df, chunk_size=10, create_advanced=True, top_features=None):\n    \"\"\"\n    Preprocess data with memory-efficient chunked lag creation and advanced features.\n    \"\"\"\n    assert len(raw_df.shape) == 2\n\n    y = raw_df['label'].to_numpy().astype(np.float32) if 'label' in raw_df.columns else None\n\n    # Original features\n    cols = [\n        'X363', 'X405', 'X321',\n        'X175', 'X179', 'X137', 'X197', 'X22', 'X40', 'X181',\n        'X28', 'X169', 'X198', 'X173',\n        'X338', 'X288', 'X385', 'X344', 'X427', 'X587', 'X450',\n        'X97', 'X52', 'X444',\n        'X598', 'X379', 'X696', 'X297', 'X138',\n        'X572', 'X343', 'X586', 'X466', 'X438', 'X452', 'X459',\n        'X435', 'X386', 'X55', 'X341', 'X683', 'X428', 'X605',\n        'X445', 'X272', 'X180', 'X593', 'X680',\n        'X686', 'X692', 'X695',\n        \"X603\", \"X674\", \"X421\", \"X333\",\n        \"X415\", \"X345\", \"X174\", \"X302\", \"X178\", \"X168\", \"X612\",\n        'X298', 'X45', 'X46', 'X39', 'X752', 'X759', 'X41', 'X42',\n        \"buy_qty\", \"sell_qty\", \"volume\",\n        \"bid_qty\", \"ask_qty\",\n    ]\n    \n    # Add new top important features\n    new_features = [\n        'X758', 'X296', 'X611', 'X780', 'X451', 'X25', 'X591',\n    ]\n    \n    # Combine all features and remove duplicates while preserving order\n    cols = list(dict.fromkeys(cols + new_features))\n    \n    # Check which features actually exist in the dataframe\n    available_cols = [col for col in cols if col in raw_df.columns]\n    missing_cols = [col for col in cols if col not in raw_df.columns]\n    \n    if missing_cols:\n        print(f\"Warning: The following features are not in the dataset: {missing_cols}\")\n    \n    print(f\"Using {len(available_cols)} features out of {len(cols)} requested\")\n\n    # Select and optimize base features\n    df = raw_df[available_cols].copy()\n    df = optimize_memory(df, verbose=True)\n    assert df.isna().sum().sum() == 0\n\n    # Create advanced features if requested\n    if create_advanced and top_features is not None:\n        df = create_advanced_features(df, top_features)\n        df = optimize_memory(df, verbose=False)\n\n    # Extended lag features\n    lag_periods = [\n        1, 3, 5, 6, 7, 8, 9,  # Very short-term (1-10)\n        12, 15, 18, 20, 30,          # Short-term (12-30)\n        120, 150, 365,\n    ]\n    \n    # Process lags in chunks to manage memory\n    print(\"Creating lagged features in chunks...\")\n    \n    # Start with base features\n    result_df = df.copy()\n    \n    # Process lags in chunks\n    for i in range(0, len(lag_periods), chunk_size):\n        chunk_lags = lag_periods[i:i+chunk_size]\n        print(f\"  Processing lags: {chunk_lags}\")\n        \n        # Create lagged features for this chunk\n        chunk_dfs = []\n        for lag in chunk_lags:\n            lagged = df.shift(-lag).add_suffix(f'_lead_{lag}')\n            lagged = lagged.fillna(0.0).astype(np.float32)\n            chunk_dfs.append(lagged)\n        \n        # Concatenate chunk\n        if chunk_dfs:\n            chunk_combined = pd.concat(chunk_dfs, axis=1)\n            result_df = pd.concat([result_df, chunk_combined], axis=1)\n            \n            # Clean up\n            del chunk_dfs, chunk_combined\n            gc.collect()\n    \n    # Final optimization\n    result_df = optimize_memory(result_df, verbose=True)\n    \n    assert 'label' not in result_df.columns\n    assert raw_df.shape[0] == result_df.shape[0] and (raw_df.index == result_df.index).all()\n    assert result_df.isna().sum().sum() == 0\n    if y is not None:\n        assert result_df.shape[0] == y.shape[0]\n    \n    print(f\"Final feature count: {result_df.shape[1]}\")\n    \n    return result_df, y\n\n# Set memory-efficient options for pandas\npd.options.mode.chained_assignment = None  # Disable SettingWithCopyWarning\npd.options.display.max_columns = None\n\n## ...existing code...\n# Load and preprocess training data\nprint(\"Loading training data...\")\ntrain_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/train.parquet')\n\n# Display available columns to verify feature existence\nprint(f\"\\nTotal columns in training data: {len(train_df.columns)}\")\nprint(f\"Sample columns: {list(train_df.columns[:20])}\")\n\n# Optimize memory for the raw training data\nprint(\"\\nOptimizing memory for raw training data...\")\ntrain_df = optimize_memory(train_df, verbose=True)\n\n# First pass: Get initial feature importance with simple model\nprint(\"\\nFirst pass: Basic preprocessing to identify top features...\")\nX_train_basic, y_train = preprocess_data_chunked(train_df, chunk_size=10, create_advanced=False)\n\n# Train basic Ridge to get feature importance\nbasic_model = Ridge(alpha=1.0)\nbasic_model.fit(X_train_basic, y_train)\n\n# Get top features for advanced feature engineering\ncoef_series = pd.Series(basic_model.coef_, index=basic_model.feature_names_in_).abs().sort_values(ascending=False)\ntop_features = coef_series.head(20).index.tolist()\n\nprint(f\"\\nTop features identified: {top_features[:10]}\")\n\n# Clean up basic features\ndel X_train_basic, basic_model\ngc.collect()\n\n# Second pass: Create advanced features\nprint(\"\\nSecond pass: Creating advanced features...\")\nX_train, y_train = preprocess_data_chunked(train_df, chunk_size=10, create_advanced=True, top_features=top_features)\n\n# Clean up training dataframe\ndel train_df\ngc.collect()\n\nprint(f\"\\nTraining data shape: X={X_train.shape}, y={y_train.shape}\")\n\n# Create ensemble models\nmodels = create_ensemble_models()\n\n# Validate models using time series cross-validation\nvalidation_results = validate_models_time_series(models, X_train, y_train, n_splits=3)\n\n# Select best performing models for ensemble\nbest_models = {}\nfor name, results in validation_results.items():\n    if results['mean'] > 0.01:  # Only include models with positive correlation\n        best_models[name] = models[name]\n\nprint(f\"\\nSelected models for ensemble: {list(best_models.keys())}\")\n\n# Create and train ensemble\nif len(best_models) > 1:\n    print(\"\\nTraining ensemble model...\")\n    ensemble = VotingRegressor(list(best_models.items()))\n    ensemble.fit(X_train, y_train)\n    final_model = ensemble\n    print(\"Using ensemble model\")\nelse:\n    # Fallback to best single model\n    best_model_name = max(validation_results.keys(), key=lambda x: validation_results[x]['mean'])\n    final_model = models[best_model_name]\n    final_model.fit(X_train, y_train)\n    print(f\"Using single model: {best_model_name}\")\n\n# Feature importance analysis\nif hasattr(final_model, 'feature_importances_'):\n    # For tree-based models\n    feature_importance = pd.Series(final_model.feature_importances_, index=X_train.columns)\nelif hasattr(final_model, 'coef_'):\n    # For linear models\n    feature_importance = pd.Series(np.abs(final_model.coef_), index=X_train.columns)\nelif hasattr(final_model, 'estimators_'):\n    # For ensemble models, average feature importance\n    importance_sum = np.zeros(X_train.shape[1])\n    valid_estimators = 0\n    \n    for estimator in final_model.estimators_:\n        if hasattr(estimator, 'feature_importances_'):\n            importance_sum += estimator.feature_importances_\n            valid_estimators += 1\n        elif hasattr(estimator, 'coef_'):\n            importance_sum += np.abs(estimator.coef_)\n            valid_estimators += 1\n    \n    if valid_estimators > 0:\n        feature_importance = pd.Series(importance_sum / valid_estimators, index=X_train.columns)\n    else:\n        feature_importance = pd.Series(np.ones(X_train.shape[1]), index=X_train.columns)\n\n# Plot feature importance\nfeature_importance = feature_importance.sort_values(ascending=False)\ntop_100_features = feature_importance.head(100)\n\nplt.figure(figsize=(12, 20))\ntop_100_features.sort_values().plot(kind='barh')\nplt.title('Top 100 Feature Importance')\nplt.xlabel('Importance')\nplt.tight_layout()\nplt.show()\n\n# Display top 20 features\nprint(\"\\nTop 20 most important features:\")\nfor i, (feat, importance) in enumerate(feature_importance.head(20).items(), 1):\n    print(f\"{i:2d}. {feat:40s} {importance:.6f}\")\n\n# Clean up training data\ndel X_train, y_train\ngc.collect()\n# ...existing code...\n\n# Load test data\nprint(\"\\nLoading test data...\")\ntest_df = pd.read_parquet('/kaggle/input/drw-crypto-market-prediction/test.parquet')\n\n# Optimize memory for test data\nprint(\"\\nOptimizing memory for raw test data...\")\ntest_df = optimize_memory(test_df, verbose=True)\n\n# Try to load precomputed timestamp reconstruction data\ntimestamp_recon_path = '/kaggle/input/the-order-of-the-test-rows-2/closest_rows.csv'\nuse_timestamp_reconstruction = os.path.exists(timestamp_recon_path)\n\nif use_timestamp_reconstruction:\n    print(\"Found timestamp reconstruction file, loading...\")\n    \n    # Load precomputed timestamp reconstruction data\n    t = pd.Series(pd.read_csv(timestamp_recon_path)['0'].to_numpy())\n    assert t.shape == (test_df.shape[0],)\n    print('Reconstructed timestamps share:', len(t[t >= 0]) / len(t))\n\n    # Visualize the reconstructed timestamps\n    plt.figure(figsize=(16, 4))\n    plt.plot(t.sort_values().to_numpy())\n    plt.title('Sorted Reconstructed Timestamps')\n    plt.show()\n\n    plt.figure(figsize=(16, 4))\n    plt.plot(t[t >= 0].sort_values().iloc[:1000].to_numpy())\n    plt.axhline(10080, color='r', linestyle='--')\n    plt.title('First 1000 Valid Reconstructed Timestamps')\n    plt.show()\n\n    # Process timestamp reconstruction\n    t -= 10080\n    t[t < 0] = 538149\n\n    t = t.sort_values()\n    t[t <= len(t)] = np.arange(t[t <= len(t)].shape[0])\n    t = t.sort_index()\n\n    t = pd.Series(np.arange(538150), index=t.to_numpy()).sort_index()\n\n    # Visualize test data before sorting\n    if 'X656' in test_df.columns:\n        plt.figure(figsize=(16, 4))\n        plt.plot(test_df['X656'].to_numpy())\n        plt.title('Test Data Feature X656 - Before Sorting')\n        plt.show()\n\n    # Sort test dataset by reconstructed time order\n    test_df = test_df.iloc[t.to_numpy()]\n\n    # Visualize test data after sorting\n    if 'X656' in test_df.columns:\n        plt.figure(figsize=(16, 4))\n        plt.plot(test_df['X656'].to_numpy())\n        plt.title('Test Data Feature X656 - After Sorting')\n        plt.show()\nelse:\n    print(\"WARNING: Timestamp reconstruction file not found!\")\n    print(f\"Expected path: {timestamp_recon_path}\")\n    print(\"Proceeding without timestamp reconstruction...\")\n    print(\"This may significantly impact model performance since lagged features assume temporal order.\")\n    \n    t = pd.Series(np.arange(len(test_df)))\n\n# Preprocess test data\n# ...existing code for loading and timestamp reconstruction...\n\n# Preprocess test data\nprint(\"\\nPreprocessing test data...\")\nX_test, _ = preprocess_data_chunked(test_df, chunk_size=10, create_advanced=True, top_features=top_features)\n\n# Clean up test dataframe\ndel test_df\ngc.collect()\n\nprint(f\"Test data shape: {X_test.shape}\")\n\n# Make predictions in batches to save memory\nprint(\"\\nMaking predictions...\")\nbatch_size = 50000  # Reduced batch size due to more features\nn_samples = X_test.shape[0]\ny_pred = np.zeros(n_samples, dtype=np.float32)\n\nfor i in range(0, n_samples, batch_size):\n    end_idx = min(i + batch_size, n_samples)\n    print(f\"  Predicting batch {i//batch_size + 1}/{(n_samples + batch_size - 1)//batch_size}\")\n    y_pred[i:end_idx] = final_model.predict(X_test.iloc[i:end_idx]).astype(np.float32)\n\n# ...rest of existing code...\n# Clean up test features\ndel X_test\ngc.collect()\n\n# Display prediction statistics\nprint(\"\\nPrediction statistics:\")\nprint(pd.Series(y_pred).describe())\n\n# Plot cumulative predictions\nplt.figure(figsize=(16, 4))\nplt.plot(np.cumsum(y_pred))\nplt.title('Cumulative Predictions')\nplt.xlabel('Sample Index')\nplt.ylabel('Cumulative Sum')\nplt.grid(True, alpha=0.3)\nplt.show()\n\n# Plot prediction distribution\nplt.figure(figsize=(12, 6))\nplt.subplot(1, 2, 1)\nplt.hist(y_pred, bins=100, alpha=0.7, edgecolor='black')\nplt.title('Prediction Distribution')\nplt.xlabel('Predicted Value')\nplt.ylabel('Frequency')\n\nplt.subplot(1, 2, 2)\nplt.plot(y_pred[:1000])\nplt.title('First 1000 Predictions')\nplt.xlabel('Sample Index')\nplt.ylabel('Predicted Value')\nplt.tight_layout()\nplt.show()\n\n# Prepare submission\nprint(\"\\nPreparing submission...\")\nsubmission = pd.read_csv('/kaggle/input/drw-crypto-market-prediction/sample_submission.csv')\n\nif use_timestamp_reconstruction:\n    # Reorder submission to match original test order\n    submission = submission.iloc[t.to_numpy()]\n    submission['prediction'] = y_pred\n    submission = submission.sort_index()\nelse:\n    # If no timestamp reconstruction, just use predictions in order\n    submission['prediction'] = y_pred\n\n# Save submission\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission saved to 'submission.csv'\")\n\n# Display submission\nprint(\"\\nSubmission preview:\")\nprint(submission.head())\nprint(f\"\\nSubmission shape: {submission.shape}\")\nprint(f\"Prediction range: [{submission['prediction'].min():.6f}, {submission['prediction'].max():.6f}]\")\n\n# Final memory cleanup\ngc.collect()\nprint(\"\\nDone!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T10:04:00.983892Z","iopub.execute_input":"2025-07-12T10:04:00.984495Z","iopub.status.idle":"2025-07-12T10:32:47.608667Z","shell.execute_reply.started":"2025-07-12T10:04:00.984470Z","shell.execute_reply":"2025-07-12T10:32:47.608075Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -U -q PyDrive\nfrom pydrive.auth import GoogleAuth\nfrom pydrive.drive import GoogleDrive\nfrom google.colab import auth  # This works in Kaggle too sometimes\nfrom oauth2client.client import GoogleCredentials\n\nauth.authenticate_user()\ngauth = GoogleAuth()\ngauth.credentials = GoogleCredentials.get_application_default()\ndrive = GoogleDrive(gauth)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T10:41:40.441649Z","iopub.execute_input":"2025-07-12T10:41:40.441921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Upload ZIP or CSV\nupload_file = drive.CreateFile({'title': 'my_csv_backup.zip'})\nupload_file.SetContentFile('/kaggle/working/my_csv_backup.zip')\nupload_file.Upload()\n\nprint(\"✅ File uploaded to Google Drive!\")\nprint(f\"🔗 Shareable Link: https://drive.google.com/file/d/{upload_file['id']}/view?usp=sharing\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-12T10:39:54.876180Z","iopub.execute_input":"2025-07-12T10:39:54.876674Z","iopub.status.idle":"2025-07-12T10:39:54.881534Z","shell.execute_reply.started":"2025-07-12T10:39:54.876649Z","shell.execute_reply":"2025-07-12T10:39:54.880917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}