{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import xgboost as xgb\nimport numpy as np\nfrom sklearn.datasets import load_breast_cancer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-10-31T15:19:13.263684Z","iopub.execute_input":"2023-10-31T15:19:13.264360Z","iopub.status.idle":"2023-10-31T15:19:13.271160Z","shell.execute_reply.started":"2023-10-31T15:19:13.264321Z","shell.execute_reply":"2023-10-31T15:19:13.268936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load a sample dataset (Breast Cancer dataset)\ndata = load_breast_cancer()\nX = data.data\ny = data.target","metadata":{"execution":{"iopub.status.busy":"2023-10-31T15:19:25.533096Z","iopub.execute_input":"2023-10-31T15:19:25.533535Z","iopub.status.idle":"2023-10-31T15:19:25.551213Z","shell.execute_reply.started":"2023-10-31T15:19:25.533499Z","shell.execute_reply":"2023-10-31T15:19:25.549676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the dataset into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Create DMatrix objects for training and testing data\ndtrain = xgb.DMatrix(X_train, label=y_train)\ndtest = xgb.DMatrix(X_test, label=y_test)","metadata":{"execution":{"iopub.status.busy":"2023-10-31T15:19:32.230542Z","iopub.execute_input":"2023-10-31T15:19:32.231017Z","iopub.status.idle":"2023-10-31T15:19:32.239938Z","shell.execute_reply.started":"2023-10-31T15:19:32.230982Z","shell.execute_reply":"2023-10-31T15:19:32.238925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set XGBoost parameters\nparams = {\n    \"objective\": \"binary:logistic\",  # Binary classification\n    \"max_depth\": 3,  # Maximum tree depth\n    \"learning_rate\": 0.1,  # Step size shrinkage\n    \"eval_metric\": \"logloss\"  # Evaluation metric to monitor during training\n}","metadata":{"execution":{"iopub.status.busy":"2023-10-31T15:19:39.981083Z","iopub.execute_input":"2023-10-31T15:19:39.981513Z","iopub.status.idle":"2023-10-31T15:19:39.987335Z","shell.execute_reply.started":"2023-10-31T15:19:39.981477Z","shell.execute_reply":"2023-10-31T15:19:39.986062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the XGBoost model\nnum_round = 100  # Number of boosting rounds\nbst = xgb.train(params, dtrain, num_round)","metadata":{"execution":{"iopub.status.busy":"2023-10-31T15:19:45.458459Z","iopub.execute_input":"2023-10-31T15:19:45.458928Z","iopub.status.idle":"2023-10-31T15:19:45.553633Z","shell.execute_reply.started":"2023-10-31T15:19:45.458892Z","shell.execute_reply":"2023-10-31T15:19:45.552562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on the test set\ny_pred = bst.predict(dtest)\ny_pred","metadata":{"execution":{"iopub.status.busy":"2023-10-31T15:20:19.533506Z","iopub.execute_input":"2023-10-31T15:20:19.533987Z","iopub.status.idle":"2023-10-31T15:20:19.550656Z","shell.execute_reply.started":"2023-10-31T15:20:19.533946Z","shell.execute_reply":"2023-10-31T15:20:19.549672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert predicted probabilities to binary predictions\ny_pred_binary = [1 if p > 0.5 else 0 for p in y_pred]\ny_pred_binary","metadata":{"execution":{"iopub.status.busy":"2023-10-31T15:20:27.935297Z","iopub.execute_input":"2023-10-31T15:20:27.935756Z","iopub.status.idle":"2023-10-31T15:20:27.946090Z","shell.execute_reply.started":"2023-10-31T15:20:27.935718Z","shell.execute_reply":"2023-10-31T15:20:27.945007Z"},"_kg_hide-output":true,"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate accuracy\naccuracy = accuracy_score(y_test, y_pred_binary)\nprint(f\"Accuracy: {accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2023-10-31T15:20:06.899658Z","iopub.execute_input":"2023-10-31T15:20:06.900058Z","iopub.status.idle":"2023-10-31T15:20:06.906879Z","shell.execute_reply.started":"2023-10-31T15:20:06.900026Z","shell.execute_reply":"2023-10-31T15:20:06.905632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You can also save and load the trained model for future use\nbst.save_model(\"xgboost_model.model\")\nloaded_model = xgb.Booster(model_file=\"xgboost_model.model\")","metadata":{"execution":{"iopub.status.busy":"2023-10-31T15:20:45.568195Z","iopub.execute_input":"2023-10-31T15:20:45.568597Z","iopub.status.idle":"2023-10-31T15:20:45.574298Z","shell.execute_reply.started":"2023-10-31T15:20:45.568565Z","shell.execute_reply":"2023-10-31T15:20:45.573336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use the loaded model for prediction\ny_pred_loaded = loaded_model.predict(dtest)\ny_pred_loaded","metadata":{"execution":{"iopub.status.busy":"2023-10-31T15:20:59.226466Z","iopub.execute_input":"2023-10-31T15:20:59.227711Z","iopub.status.idle":"2023-10-31T15:20:59.247002Z","shell.execute_reply.started":"2023-10-31T15:20:59.227658Z","shell.execute_reply":"2023-10-31T15:20:59.246136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}