{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":118152,"databundleVersionId":14157350,"sourceType":"competition"},{"sourceId":13414258,"sourceType":"datasetVersion","datasetId":8513584},{"sourceId":611428,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":459277,"modelId":475154}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# CARE Score and Care2CompareDataset usage\n\nThis notebook shows how to apply a fault detection model to the CARE2Compare [dataset](https://doi.org/10.5281/zenodo.10958774) using the EnergyFaultDetector package as an example. For evaluation the kaggle version of the CARE-Score is used.\nFor more indepth information of how to customize your model evaluation using the CARE classes from the EnergyFaultDetector evaluation module, refer to the [CARE to compare notebook](https://github.com/AEFDI/EnergyFaultDetector/blob/main/notebooks/CARE%20to%20Compare.ipynb).\nContents of this notebook:\n\n1. Using the Care2CompareDataset class to load the CARE to compare dataset from zenodo.\n2. Applying an EnergyFaultDetector model to the dataset.\n3. Saving the submission in the correct format.","metadata":{}},{"cell_type":"code","source":"!pip install energy_fault_detector --upgrade","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nimport pandas as pd\n\nfrom energy_fault_detector.fault_detector import FaultDetector\nfrom energy_fault_detector.config import Config\nfrom energy_fault_detector.evaluation import Care2CompareDataset","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = Path('../input/caredataset/CARE_To_Compare')  # TODO: insert path to the Care to compare main directory or set download_dataset to True\nc2c = Care2CompareDataset(path=data_dir, download_dataset=False)  # If you have already downloaded the dataset, set download_dataset to False an specify the datapath","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"index_column = 'id'  # time_stamp as index column is also possible\n\n# prepare submission DataFrame\nevent_ids = c2c.event_info_all[\"event_id\"]\nnum_events = len(event_ids)\nsubmission_dataframe = pd.DataFrame(\n    index=event_ids.values,  # needed for easy result saving\n    data={\"event_id\": list(event_ids.values),  # IMPORTANT: Add the event_ids as a list to the DataFrame. This is needed for kaggle evaluation format!\n          \"prediction\": num_events * [None],  # This column must contain the boolean prediction value for each event.\n          \"prediction_index\": num_events * [None]  # This column must contain the index list for each prediction of each event.\n    }\n)\nconfig_base_path = Path('../input/c2c_configs/other/default/1/c2c_configs')\n# Load EnergyFaultDetector configs\nconfigs = {\n    'A': Config(config_base_path / 'windfarm_A.yaml'),\n    'B': Config(config_base_path / 'windfarm_B.yaml'),\n    'C': Config(config_base_path / 'windfarm_C.yaml'),\n}\n\nfor wf in ['A', 'B', 'C']:\n    print('Evaluating wind Farm ', wf)\n    # update EnergyFaultDetector model config depending on the feature description of the current wind farm\n    config = configs[wf]\n    c2c.update_c2c_config(config, wf)\n\n    wf_event_ids = c2c.event_info_all.loc[c2c.event_info_all['wind_farm'] == wf, 'event_id']\n    for event_id in wf_event_ids:\n        print(\"Evaluating Event\", event_id)\n        x_train, x_test = c2c.load_event_dataset(event_id=event_id, index_column=\"id\")\n\n        # create normal index\n        y_train = x_train['status_type_id'] == 0\n        \n        # drop non sensor data features\n        x_train = x_train.drop(['asset_id', 'status_type_id', 'time_stamp'], axis=1, errors='ignore')\n        x_test = x_test.drop(['asset_id', 'status_type_id', 'time_stamp'], axis=1, errors='ignore')\n        \n        # create model\n        model = FaultDetector(config)  # TODO: insert your model\n        # train and predict\n        train_results = model.fit(sensor_data=x_train, normal_index=y_train, save_models=False)  # TODO: make sure your model can handle x_train and y_train as input parameters\n        prediction = model.predict(x_test).predicted_anomalies  # TODO: prediction must be pandas.Series containing booleans (same index as x_test)\n\n        # Add the prediction to the submission DataFrame\n        submission_dataframe.at[event_id, \"prediction\"] = list(prediction)\n        submission_dataframe.at[event_id,\"prediction_index\"] = list(prediction.index)\n\n# save submission file in csv-format\nsubmission_dataframe.to_csv(\"./submission.csv\", index=False)  # Important: Omit the index and use ',' as separator since kaggle requires it!","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_dataframe","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}