{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118152,"databundleVersionId":14157350,"sourceType":"competition"}],"dockerImageVersionId":31259,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-02-06T03:32:05.584142Z","iopub.execute_input":"2026-02-06T03:32:05.584405Z","iopub.status.idle":"2026-02-06T03:32:07.062634Z","shell.execute_reply.started":"2026-02-06T03:32:05.584370Z","shell.execute_reply":"2026-02-06T03:32:07.061789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport ast\nfrom sklearn.ensemble import IsolationForest\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n\n# ==========================================\n# CONFIGURATION\n# ==========================================\n# Path to the dataset root directory\nDATA_ROOT = '/kaggle/input/we-do-wind-energy-fault-detector-challenge/CARE_To_Compare/CARE_To_Compare'\nSAMPLE_SUB_PATH = '/kaggle/input/we-do-wind-energy-fault-detector-challenge/sample_submission.csv'\n\n# Contamination: expected proportion of outliers. \n# Keep this LOW (e.g., 0.005) to be conservative and avoid False Positives.\nCONTAMINATION = 0.005 \n\n# ==========================================\n# HELPER FUNCTIONS\n# ==========================================\n\ndef find_dataset_file(event_id, root_dir):\n    \"\"\"Finds the .csv file for a specific event_id within the subdirectories.\"\"\"\n    filename = f\"{event_id}.csv\"\n    for subdir, dirs, files in os.walk(root_dir):\n        if filename in files:\n            return os.path.join(subdir, filename)\n    return None\n\ndef train_and_predict(event_id, prediction_indices):\n    \"\"\"\n    Loads data, trains Isolation Forest, and predicts anomalies for the requested indices.\n    \"\"\"\n    # 1. Find and Load Data\n    file_path = find_dataset_file(event_id, DATA_ROOT)\n    \n    if file_path is None:\n        # Fallback: If file not found, predict all False (Safe)\n        return [False] * len(prediction_indices)\n    \n    try:\n        df = pd.read_csv(file_path)\n        \n        # 2. Preprocessing\n        # Select numeric columns only for training\n        feature_cols = df.select_dtypes(include=[np.number]).columns\n        \n        # Handle missing values (impute with mean)\n        imputer = SimpleImputer(strategy='mean')\n        X = imputer.fit_transform(df[feature_cols])\n        \n        # Scale features (helps Isolation Forest)\n        scaler = StandardScaler()\n        X_scaled = scaler.fit_transform(X)\n        \n        # 3. Train Isolation Forest\n        # n_jobs=-1 uses all CPUs for speed\n        iso_forest = IsolationForest(contamination=CONTAMINATION, n_jobs=-1, random_state=42)\n        iso_forest.fit(X_scaled)\n        \n        # 4. Predict\n        # We need predictions only for the specific indices requested in sample_submission\n        # We get the anomaly labels for the whole dataset first, then slice.\n        # IsolationForest returns -1 for outlier, 1 for inlier\n        full_predictions = iso_forest.predict(X_scaled)\n        \n        # Convert to Boolean: -1 (outlier) -> True, 1 (inlier) -> False\n        is_anomaly = (full_predictions == -1)\n        \n        # Extract predictions for the specific required indices\n        # Assuming prediction_indices correspond to the DataFrame's row index (0 to N)\n        # We must ensure indices are within bounds\n        max_idx = len(df) - 1\n        valid_indices = [i for i in prediction_indices if i <= max_idx]\n        \n        # Get results\n        results = is_anomaly[valid_indices]\n        \n        # If any indices were out of bounds, pad with False\n        if len(results) < len(prediction_indices):\n            pad_len = len(prediction_indices) - len(results)\n            results = np.concatenate([results, [False]*pad_len])\n            \n        return results.tolist()\n        \n    except Exception as e:\n        print(f\"Error processing event {event_id}: {e}\")\n        # Fallback to All False on error\n        return [False] * len(prediction_indices)\n\n# ==========================================\n# MAIN EXECUTION\n# ==========================================\n\nprint(\"Loading sample submission...\")\nsubmission_df = pd.read_csv(SAMPLE_SUB_PATH)\n\npredictions = []\n\nprint(f\"Processing {len(submission_df)} events...\")\n\nfor idx, row in submission_df.iterrows():\n    event_id = row['event_id']\n    \n    # Parse prediction_index string into a list of integers\n    if isinstance(row['prediction_index'], str):\n        indices = ast.literal_eval(row['prediction_index'])\n    else:\n        indices = row['prediction_index']\n        \n    # Generate predictions\n    pred_values = train_and_predict(event_id, indices)\n    \n    # Store as string representation of list (e.g., \"[True, False, ...]\")\n    predictions.append(str(pred_values))\n    \n    if idx % 10 == 0:\n        print(f\"Processed {idx}/{len(submission_df)} events\")\n\n# Assign predictions back to dataframe\nsubmission_df['prediction'] = predictions\n\n# Save submission\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"Done! Saved to submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-06T03:36:28.356778Z","iopub.execute_input":"2026-02-06T03:36:28.357054Z","iopub.status.idle":"2026-02-06T03:41:30.332988Z","shell.execute_reply.started":"2026-02-06T03:36:28.357031Z","shell.execute_reply":"2026-02-06T03:41:30.331043Z"}},"outputs":[],"execution_count":null}]}