{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#!/usr/bin/env python3\n\"\"\"\nSimple runner script for the comprehensive feature selection pipeline.\nThis script handles package installation and provides a clean interface.\n\"\"\"\n\nimport subprocess\nimport sys\nimport os\n\ndef install_packages():\n    \"\"\"Install required packages\"\"\"\n    packages = [\n        \"pandas\",\n        \"numpy\", \n        \"scikit-learn\",\n        \"xgboost\",\n        \"lightgbm\",\n        \"tensorflow\",\n        \"shap\",\n        \n        # Optional packages (will try to install but continue if fail)\n        \"cudf\",\n        \"cuml\", \n        \"featurewiz\",\n        \"mrmr\",\n        \"boruta\",\n        \"probatus\",\n        \"tsfresh\",\n        \"feature-engine\",\n        \"mlxtend\",\n        \"h2o\"\n    ]\n    \n    print(\"Installing required packages...\")\n    \n    for package in packages[:8]:  # Core packages\n        print(f\"Installing {package}...\")\n        try:\n            subprocess.check_call([sys.executable, \"-m\", \"pip\", \"install\", package, \"-q\"])\n        except:\n            print(f\"Failed to install {package}\")\n    \n    print(\"\\nInstalling optional packages (failures are OK)...\")\n    for package in packages[8:]:  # Optional packages\n        print(f\"Trying to install {package}...\")\n        try:\n            subprocess.check_call([sys.executable, \"-m\", \"pip\", \"install\", package, \"-q\"])\n            print(f\"✓ {package} installed\")\n        except:\n            print(f\"✗ {package} not available (will use fallback)\")\n\ndef run_feature_selection():\n    \"\"\"Run the feature selection pipeline\"\"\"\n    \n    # Import the main pipeline\n    from comprehensive_feature_selection import main, Config\n    \n    print(\"\\n\" + \"=\"*80)\n    print(\"STARTING FEATURE SELECTION PIPELINE\")\n    print(\"=\"*80)\n    \n    # Check if resuming\n    if os.path.exists(Config.progress_file):\n        print(\"\\n⚠️  Found existing progress file!\")\n        response = input(\"Resume from previous run? (y/n): \")\n        if response.lower() != 'y':\n            os.remove(Config.progress_file)\n            print(\"Starting fresh run...\")\n    \n    # Run the pipeline\n    try:\n        results = main()\n        \n        print(\"\\n\" + \"=\"*80)\n        print(\"PIPELINE COMPLETED SUCCESSFULLY\")\n        print(\"=\"*80)\n        \n        # Show quick summary\n        completed = sum(1 for r in results.values() if r.get('selected_features'))\n        print(f\"\\nMethods completed: {completed}/{len(results)}\")\n        \n        # Show top features\n        all_features = []\n        for result in results.values():\n            if result.get('selected_features'):\n                all_features.extend(result['selected_features'][:10])\n        \n        from collections import Counter\n        top_features = Counter(all_features).most_common(10)\n        \n        print(\"\\nTop 10 most selected features:\")\n        for i, (feature, count) in enumerate(top_features, 1):\n            print(f\"{i:2d}. {feature}: selected by {count} methods\")\n        \n        print(f\"\\nDetailed results saved in: {Config.results_dir}/\")\n        \n    except KeyboardInterrupt:\n        print(\"\\n\\n⚠️  Pipeline interrupted by user\")\n        print(\"Progress has been saved. Run again to resume.\")\n    except Exception as e:\n        print(f\"\\n\\n❌ Pipeline failed with error: {str(e)}\")\n        print(\"Progress has been saved. Check the progress file for details.\")\n        raise\n\ndef main():\n    \"\"\"Main entry point\"\"\"\n    \n    # Check if running in Kaggle\n    if os.path.exists('/kaggle/input'):\n        print(\"Detected Kaggle environment\")\n    else:\n        print(\"⚠️  Warning: Not running in Kaggle environment\")\n        print(\"Make sure data paths in Config are correct\")\n    \n    # Option to install packages\n    response = input(\"\\nInstall/update packages? (y/n): \")\n    if response.lower() == 'y':\n        install_packages()\n    \n    # Run feature selection\n    run_feature_selection()\n\nif __name__ == \"__main__\":\n    main()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}