{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install Rbeast","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:02:57.469562Z","iopub.execute_input":"2024-11-27T17:02:57.471277Z","iopub.status.idle":"2024-11-27T17:03:09.037868Z","shell.execute_reply.started":"2024-11-27T17:02:57.471220Z","shell.execute_reply":"2024-11-27T17:03:09.036654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport pyarrow.parquet as pq\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib.ticker import PercentFormatter\nfrom matplotlib.ticker import MaxNLocator\nimport seaborn as sns\nimport Rbeast as rb;\nimport os\nfrom concurrent.futures import ThreadPoolExecutor\nfrom tqdm import tqdm\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.cluster import KMeans\nfrom sklearn.decomposition import PCA\nfrom matplotlib import pyplot\nfrom sklearn.metrics import silhouette_samples, silhouette_score\nfrom sklearn.cluster import AgglomerativeClustering\nfrom imblearn.over_sampling import SMOTE\nfrom collections import Counter\nfrom sklearn.metrics import classification_report, confusion_matrix, ConfusionMatrixDisplay\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import cross_validate\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import PolynomialFeatures\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.linear_model import Ridge\nimport xgboost as xgb\nfrom xgboost import XGBClassifier","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:09.040318Z","iopub.execute_input":"2024-11-27T17:03:09.041144Z","iopub.status.idle":"2024-11-27T17:03:10.858992Z","shell.execute_reply.started":"2024-11-27T17:03:09.041092Z","shell.execute_reply":"2024-11-27T17:03:10.858139Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndf.head(30)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:10.860159Z","iopub.execute_input":"2024-11-27T17:03:10.860715Z","iopub.status.idle":"2024-11-27T17:03:10.999653Z","shell.execute_reply.started":"2024-11-27T17:03:10.860670Z","shell.execute_reply":"2024-11-27T17:03:10.998761Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"NORMALIZAR","metadata":{}},{"cell_type":"code","source":"copy_df = df.copy()\n#copy_df.drop('id', axis=1, inplace=True)\n\n# Tokenizar variables categóricas usando OneHotEncoder\ncategorical_columns = copy_df.select_dtypes(include=['object']).columns\nonehot_encoder = OneHotEncoder(sparse_output=False, drop='first')\n\n# Aplicar OneHotEncoder a las columnas categóricas\nencoded_columns = pd.DataFrame(onehot_encoder.fit_transform(copy_df[categorical_columns]))\n\n# Obtener los nombres de las columnas codificadas\nencoded_columns.columns = onehot_encoder.get_feature_names_out(categorical_columns)\n\n# Reemplazar las columnas categóricas originales por las codificadas\ncopy_df = copy_df.drop(columns=categorical_columns).reset_index(drop=True)\ncopy_df = pd.concat([copy_df, encoded_columns], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:11.001741Z","iopub.execute_input":"2024-11-27T17:03:11.002056Z","iopub.status.idle":"2024-11-27T17:03:11.354791Z","shell.execute_reply.started":"2024-11-27T17:03:11.002027Z","shell.execute_reply":"2024-11-27T17:03:11.353928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Normalizar los datos usando StandardScaler (media=0, desviación estándar=1)\nscaler = StandardScaler()\ncopy_df_scaled = pd.DataFrame(scaler.fit_transform(copy_df), columns=copy_df.columns)\ndf_normalizado = copy_df_scaled;\n\n# Verificar los primeros registros\nprint(df_normalizado.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:11.356068Z","iopub.execute_input":"2024-11-27T17:03:11.356545Z","iopub.status.idle":"2024-11-27T17:03:12.062636Z","shell.execute_reply.started":"2024-11-27T17:03:11.356468Z","shell.execute_reply":"2024-11-27T17:03:12.061583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def mostrar_dispersion(df):\n    # Seleccionar solo las columnas numéricas\n    columnas_numericas = df.select_dtypes(include=['number']).columns.tolist()\n    \n    # Verificar que haya columnas numéricas\n    if len(columnas_numericas) == 0:\n        print(\"No hay columnas numéricas para graficar.\")\n        return\n    \n    # Graficar scatter plots (pairplot) para las columnas numéricas\n    sns.pairplot(data=df, vars=columnas_numericas)\n    \n    # Título del gráfico\n    plt.suptitle(f'Scatter Plots de Columnas Numéricas', y=1.02)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:12.063709Z","iopub.execute_input":"2024-11-27T17:03:12.063996Z","iopub.status.idle":"2024-11-27T17:03:12.069369Z","shell.execute_reply.started":"2024-11-27T17:03:12.063968Z","shell.execute_reply":"2024-11-27T17:03:12.068361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_actigraphy(id):\n    actigraphy = pd.read_parquet(f'/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={id}/part-0.parquet')\n    data = df.loc[df.id == id]\n    age = data['Basic_Demos-Age'].iloc[0]\n    sex = ['boy', 'girl'][data['Basic_Demos-Sex'].iloc[0]]\n    \n    #diferencia de tiempo entre registros consecutivos en segundos.\n    day = actigraphy['relative_date_PCIAT'] + actigraphy['time_of_day'] / 86400e9\n    interval_seconds = day.diff() * 86400\n    \n    #asignar la nueva columna a la actigrafia para analizar cambios en el tiempo\n    actigraphy['interval_seconds'] = interval_seconds\n    \n    #remover la informacion donde no se usa el reloj (no se registran muchos datos)\n    actigraphy = actigraphy.loc[actigraphy['non-wear_flag'] == 0]\n    \n    # Caracteristicas a evaluar\n    timelines = [\n        ('X', 'm'),\n        ('Y', 'm'),\n        ('Z', 'm'),\n        ('enmo', 'forestgreen'),\n        ('anglez', 'lightblue'),\n        ('light', 'orange'),\n        ('non-wear_flag', 'chocolate'),\n        ('interval_seconds', 'k')\n    ]\n    \n    _, axs = plt.subplots(len(timelines), 1, sharex=True, figsize=(12, len(timelines) * 1.1 + 0.5))\n    for ax, (feature, color) in zip(axs, timelines):\n        ax.set_facecolor('#eeeeee')\n        ax.scatter(day.to_numpy(),\n                   actigraphy[feature].to_numpy(),\n                   color=color, label=feature, s=1\n                  )\n        ax.legend(loc='upper left', facecolor='#eeeeee')            \n           \n    axs[-1].set_xlabel('day')\n    axs[-1].xaxis.set_major_locator(MaxNLocator(integer=True))\n    plt.tight_layout()\n    axs[0].set_title(f'id={id}, {sex}, age={age}')\n    plt.show()\n    \nread_actigraphy('0417c91e')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:12.070672Z","iopub.execute_input":"2024-11-27T17:03:12.071009Z","iopub.status.idle":"2024-11-27T17:03:15.207341Z","shell.execute_reply.started":"2024-11-27T17:03:12.070981Z","shell.execute_reply":"2024-11-27T17:03:15.206307Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **FASE 3 - PREPARACIÓN DE DATOS, CREACIÓN DE VISTA MINABLE**","metadata":{}},{"cell_type":"markdown","source":"Esta fase debe ser la complementación de la fase 2, esto debido a que el solo entender los datos no es suficiente, se deben preparar los datos de cada participante y buscar una forma de integrarlos con el dataset principal, puede que haya otros enfoques para abordar este problema, pero es el que se usará para nuestro caso.\n\nPara unir los datos de cada participante en el dataset principal crearemos datos generales que nos sirvan de indicadores para cada archivo parquet de cada paciente, estos datos generales se basaran en estadisticas descriptivas que nos dan una idea general del comportamiento actigráfico de cada paciente, además crearemos variables compuestas que nos ayudaran a tener más información de cada archivo parquet.\n\nEn definitiva intentaremos obtener la siguiente información que se plasmará al final en nuestra vista minable del dataset principal: \n\n**Para cada variable dentro de la actigrafía**\n* count\n* mean\n* std\n* min\n* 25%\n* 50%\n* 75%\n* max\n\n**Variables generadas**\n* computer_time: Tiempo promedio gastado en el computador, variable compuesta por el angulo del reloj, la luz obtenida reflectada en el y el nivel de actividad física del participante\n* Combinaciones para cada tipo de instrumento de medida, esto para tener una tendencia general de cada categoría generada, para reducir y precisar la información\n    * Average_Grip_Strength\n    * Average_Sit_Reach\n    * Water_to_Lean_Mass_Ratio\n    * Metabolic_Efficiency\n    * FGC-CurlUp_Status\n    * FGC-PushlUp_Status\n    * Total_Water_Composition\n    * Lean_body_mass\n    * Muscle_and_Bone_Mass","metadata":{}},{"cell_type":"markdown","source":"![image.png](attachment:6cc5bb6e-c74a-42b4-a36e-383dbf9076b3.png)","metadata":{},"attachments":{"6cc5bb6e-c74a-42b4-a36e-383dbf9076b3.png":{"image/png":"iVBORw0KGgoAAAANSUhEUgAAAk0AAAMxCAYAAAD2bYdWAAAAAXNSR0IArs4c6QAAAARnQU1BAACxjwv8YQUAAAAJcEhZcwAADsMAAA7DAcdvqGQAAFVySURBVHhe7d1taCNZnuf7vy69kAVzIQtmIQt6uC3jhAxjw8j0wFTT/aKymRcjYUO6qIWSsGGuuhZmq2egx7JhsnLqRd/cHLDlGdit3gtzy3fBRhroJJ1gIw3speu+6KFqoRfrQhorwcYa2IUq2IEqmIYq6Ia450SckCL05L9t+TG/nyLSUoQUiifp/OKcE1Ep3xAAAAAM9b+4vwAAABiC0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwC88r6Rr774Sr75rXuKa+mbr76Qr37tnuBcEJpwrPpCSlKpgtTdcwA3zD/8WF5/43V5/S8+cyNw/dTlx6+/Ia//6x8Le/H83NDQVJdCyhT0U2vScmOulMM1mbDLt0AMwStihMd8a3Xilfru2JOWidXR/JINnNddTzLmj5e+Ez6/Sv65Iesf5mT69143J2/2BM4Or8kb3g/lR6tb0vhn97pXnifed82fybRcwb14Y1DTBOAa+UI++Yeme/wq+EzqT93DMxsyr/FF2fV92f1J2o24Gr74+7fljX89LT/6P+rS+HVasu8VpWiHGU/knz6R9eWyfPqNe/ErLy2Lv/LF/9WieYTzQmgCcG1880VNtv5f9+QV8M0/PpP6iELBKOd1Eb7aKYiX35Ivbt+X8i+/FP/LXan93cfysR22d+Xzr335+vO6FL/t3gBcgFc+NAVV/akJWTt0IxJasjaVbOZrPBwLqoYLz79yYzq+el6Q11IpmfjrEZ8J/9acXX/4toy95qqm/9XrMr3wM/ksqpY2Py52/MAq/N/2b678am9dfvQHnSrv1+7+UEpPW8KJGy5K0F9utirf/PYr+Wz1bZl43R3j5js2dr8kW//DvdD47MM35LU3fhT2rdvMtY/bnmb4474vkaDJ8A159N/M468+k7W5seD7G/89CJbv/rp8Yea59Rc/TM7z325Jq+fLEv5mvPGwYR6H6xS9J/H9/Kop6/92Wl7/V25+r43JD/8iPr8vpDr3mrz2g3Ddmsv2d8e9Nt40+c0X0vjPJfmhZ7ZNNP21N8z6rkvzJPNyzad9f0O+aSXXPdo3h31+KYLfopxUf21W8R/X5G2v6/fl+Rfuhcf4dV3e/zdVswUzUv7sF7L4/dtuQtKtO7fllnsc+eoffyaF2O9a6vVpKfzHz+Srnk7u4b567c8/C9cxvj/Me34ULWv3vjf76u3V3vmd/FgxVPvPiW1b+R9bsd/uqL9pb3kVMMf2zxam5Y32/kvJ6783LbkPq9Ls7jR+Gfv6uvFvpJqfN6smk2X/yI0Z5GjF80U8v3zgRiQc+eXJrvn8Ztf/4Dtm3K28/+xLN8768pmfv2Vf+9jf/40bN8hB2ffs8s3X3IghfrPvl79rXmte782X/crPn/kf/eRN/459/+28X7PL8Jtf+O8Hnz1gfbfzwfvf/A+fuxFmcc048zPky503/cX/9Mx/tlH28/fCz8ms7LtXhWrzdrz5LPccOLEBx3xwbN0p+h/M3w6OxfxffeQ/+/lH/gcz9ntppkXHuPH5r8xx+vNF/007/vuL5rF9bob/su9/Hb5E932JuGXKV2t+8baZ/u37fv69ol98r+x/+i/hS4Llu5f383aeZnpxpRJ8V4rfvxMu351F/9PE9939ZsxX/NqfmHX61h3//rt2nkW//Eu3lF+a3yf7eXLHf/Mndn0rfnnere93y+7342t//7+YdfsP+WDZ77xrX+fW91ed77H/4gM/bae/lfc/sN9jO6930uG83vrYD1+pmJfbFt5K1y9Ie1lv+ZmZD/yP7Pb8q6yfsb83ctvPb8c3qBH81tzxi39pf1/M+r3beY/3rQHv6ePzv7sfrEP6r5K/RcfZX8kE70vsq7fcvmpv24jbV2a9PjD79873w21YWSn6b94Jl7W4/Wl4PN3L+ot2fv/pAz/rfidv/9mnbj6hkx8rhmr/OcG2NWXVdtnPmOm3fj8bHFfFn1Tc736f8upf3LH9Lc/PBt8t8xlmHfJvpf1bt9/3fxFfnkva19cNoemkocn6VXig3/6Tmvux/jr8gTTzedwMRgx3gtC0/1f2C3TLz291HXxuGW6Zedhl+MWf3hq4HmHoedP/6L+7Ef9S8eeC7dMd8L70KzP2tcn5EJpwZsNCkx3fU6CZo7GaDabd/7t40eG+2wO+O9rvS8At0+3bt/3b88/8L7sLNKO9fDMf90yPCuhkwe5+M+w8TUh79j/d6Lav/coDO8/e34pofRPBZVCYafva/9oFvI4v/Wfv2t+DjF9uuVHWsHn1ndb5PXj8or3VQl/v+4/tekrWr8Q3tTtBCz67+7fwy4qftdO6w0Afz96187jjf/ArN0Ij2sfv9u7LL3+e92+Zaem/3HVjLLevesYb0bKa4badnxsdiE6cpZj4TTz5sWKdYP8F2/aWOa7Mie2T2IlCW5/yait8z/ufuOcDXd6+vm4ITacJTcbuX9ofZ3MmUjcHfb0Y1Np45kBWGVCA9Po0rEH6w4/6HHif+x+/ZabZMGO/oL98P/hR6PlB/I3bFrGDNzyLG/BF6jMfQhPObMAxHx5bA45FE+6DH97Ee4aFphN8X6xomW51nXHHtJfvl25EXFTDe+cDv1PkRgXxgHX6/GP/vvnMW3/6Czcizi1//PdmWNAZ4msXwPLbboQ1bF79prlx/ZfV+KTPb44rSPu/xwTGoGA+7rck2oYn+80J91Xs5DDhc/+jP3TzbO/r6HPu+x/3HDDRsvYPbrt/aWuP+p1cnuRYGazv/otCSt/j2zpDaLq0fX390BH8lDI/fSYffOcrWc9Py3R+Xb6afCzPlsxhN0qHn8on35iv7Vvf63MJ6R15I+gA2ZDmP5k/3y/KonlRc2Mr0Z79zdN1qZq/2fcK7Xns/vIT8+99+d4fhM8T7qSDKy+aL16lK5RwuQYci79zW/r3ZBngJN+XmFvmu3H/W+5JX2b5ft89jPvWfbn/wPz94jPZ7e6+casohbfc47hffSLBt+8H3wufJ9yR9Lj5s7crZ/323TLb7sxehMvRf1mNH9yXOfOn+cvPevpB9n/PLbk9gsXqrymNX5k/d8y+6tsx/I7cn7W/z5/Ip/9fOKbtzpsy3XPARMt6X6b77Pvbr7/uHnU7xbHSx7D9d/9/f7vP8T3AHxWkePsb+dn9MXn7P34iX/TpmhS4Vvv6chGaTutbGXn89ANJf9WU5lde8Ngb+sN7Cs3wQP7ir6fbHeziQ24zfFkoI4X3TNzZ+5ms77lR5vDeerpl/s5JYcacJwRa0rR9VO29rP7X3nmm7pbO/IMNnIwJR7/jHp7Fib4vHenfO64IGrx86e/Y937eexfm8f73ymm9DL58Us+/1mcZx6TU/u4q/fYbaf7DmpRmp+WN+H2MZu2p0tm0DsNlvf070W9Hl2+58a3PzRZIGviec9OS/Zfmz+++PjBo3wp+n7+wi5s05D1W+D6tEx4rp9h/d+6oI5M58cjKxy+eSfFeS7b+/IfyxmuvyfTCmtS7OnZfr319uQhNZ/DVQcPV6jTl00bv1XSjcufdj+TZz58NGD6SrPsOef8mL2mzRNWfu9jz6y159tz8ffC2zPV8kd+Uxb7zc8OfT7vXAdeL9vtymd5c6rds0VAS1bfvt01Z+8FrMpEtmVOgrDz6m3Wp/epIPv/cFGwbtl7gunpD3rDV3dKQhg1CN9VF7b9vz8nHTV++btakPJ+WxmZJcnfN5/7FJ3J+pdbNRWga6pvB/y+mr6pSyNdF3vpAPnjLnjkWpDrqI/DuhAQNft/5nsy9MzdguC9eFIgmi/L+pDkZ2KwEZ9zf7DwTW8+U/5N87LLctKSn7N/XZfqP+83PDd+9AiULcBIn/b6ofSGfD2hSaf2TnfDGwNqFbulxe99t8+3LZPssWzRkVM0vrb99W0r/VSRb/VL2tx/L++a92e+mg5qIOyM4+4+W9atfD2jT+a0bnzYhJ3w0IrfkzR/YPdmU2v+jvWw9LRP3zJ9//nJgEAh/y+/YxT1H+mPlvPdft1v3srK4sS/+//xUPvhDs3X/NieP3D3PLm9fXz+EpqDasSm7/dqkft2QT/ue6Xwh63P23hgZKf/dY3m88bHct81dC/a+IiM0/qbcN4v3xd8/C0LQ8dIyt2B+bP5pS+qHIp/8QxCZpPDH4dTI99580/xbl2c7A74gwHV04u+L1qfS6DfD334in9ia3L59Ygb4g+9J8O17unXm+6GF/Q49uf8HvY1LjV/ZnlNn9OZ987tmfkd++Wn4vNsvPwlOyrwfvBk7KRuN9LuLwWd/trwoW6ofVbMd/sju/E/k09i9vTq+kE+27fYa0OdoZPTHyrnvv0F+9015/J8fm0/+Rj75b64H7CXu6+vmlQ9NafMjZg+CrX/oPki/kc9++shEi15f/F8F+ZFJ6Jm/eSaLtuPmt4vy8d+YpL5TkMLfjzA2feu+vL+UNiHo30vhr5t9f2S/+SY5Nv3u++ZHuSnrT9eksmnOLv60KNmuNvk7f/LjoFPf1r8ryla//2+TOavg/3aOq+o1+48pmXpuw3iK74vON/Kz5TVpdn0nmn9bkp+Z2aXfKwT/3zYV81vxY9sh+Pn7Uuxzg1wrsYjmu2t/n/pdmPH6bTvlS/m8+zv8ck2Kf92numPIvPq6U5DF+Vvyzf/5Y/n3e13bzTYtLf/MbJn7svhu0JY2WmY7fbRituo3VXnbe1vW+91cscubf26DwGdSWt7qufHkVzuL8ui/itz+sx/3/B6Olv5YOfH+G6WvvjSfLPLG7eDbdLn7+rpxV9HdMO6y5FsZfy64WV3v8LgeXbR51L4Z3p3vLwY35wpuYPb7t4Kb4eX/uOsSzv8eXjJs7ysTu/jSiObTdS+LftzlnfKd+32XzQ4fN9xr7c36gktlzXAv6xftDcrM8hXfy/qZO2YZey69dpfW3rrl3xpyqemX28Xwhn/BTckW/fJGxS//pOjn3/L829/iPk0YseiY73vLgUHHVv/bC3z6Z/aeaOb7+scfBDevrKxUOpdxn+T70u8y+y7B8v1u1s/Gb1j484/8xeiGhYNumDjsdidf1vxicPPE8IaK9qaJlZVFv/jufd+73b080a0SzPj3wpsTfvR3tfCS81++H96g1nxf88Fyme/we+GNPDMPssHfxCXrw+Y1aFu0l9X8TrwX3iw0fvPH/jc87P7cjpP9lnzp/+LP3I0ezXBrPLr5qB3M/vy2OQ66tvP+37wZvt7s+/BmkbEbpZ5wXw1b1n63qjnxsXLS/XfMtu27PvX3/TvfNd+Dn4T7Lv4Zcrvo1+L3iLrUfX193OzQNGRI/Dh8feQ/ey/j37H30bDTv3Xbz8x/5H9qjpH9J+bL0T4Iox+dPjfzsprhnVplppK8GVq3qAAZMiQOxN986e/+34v+/XF7wzP3GrOM3ltFv/xJFP46orvp2vvPJO9Zm/T1Qc0vz2dMSOp87q07GT/7VxV/P/ZluqkHPy7QCEOTDUYVc9ya8/TwuL296CfuFKP9vmhDk12+33zuP/vJfT8d/41475l/1HOHQUVossxvTm0l72eCOzC74dad4E7MlWbXTL/8hf9BdFdrM9z6o4/a8/6y/kFiPW2wWNwy6+jucdVToA2a17BtYX8f4+tuTsbSby36H7/o8yt3DgXp181nJviY3+f4tjLLcOfb5rfqbz7tucnjl7/8yM9/NwzWwXA7Y0LJL/zPe+7FdQ6h6UTHygn332lC00HFL5oT4XbZZofbnn//Jx/7+/0KqUve19dByv5jVhwA0MX+/8Rym3mp+RUxhRgwEMfKq4GO4AAAAAqEJgAAAAVCEwAAgAKhCQAAQIGO4AAAAArUNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQOFGh6alpaVgAAAAOCtqmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJuBC1aWQSklqak1abkzIjQ+GCVk7dKMTBr13uNbqhKQW6u4ZAOC0CE0jFBROplAr7LgRuFGi/dsznDDE9JeViu+L79ck78ZcnJasTXHcAsBxCE0jERY6OSleQoGHi5Je2jehphNs8tv2sRleLEo6fImCC0cnek/kLO8FAJwVoWkEWqs52X3iy/6S58bglbNT6FvjFG8aS9RUnbC57Nj3Hq7JRDS9PRQk/sr6QmdaVKsUjhuT0p5IdbYzfWL17HVnAHDTEJpGwNZAVGbcE7yavGnx9nal6Z72E9VUHa2cPFwPf29L1h6URFaOwpqvbVsP5kn5oCLZ8AUimzkpTYXT7TyqD8OAl92wtWVHUp6M1ZyZYX+JuiwA6EZoAkZh3JOMexivdWq+aIo3dc41kIdbsr7nSfGBCzozBcmb+LYbT3DztXYQSj8oHhvwAAC9CE3ASHgyPdmQ5qFI/alIPtMJJZl751xrEwS2pqw/d01qOxWpmthUoPYTAEaK0ASMRFq8jK3dqUulMS2P3xGp7LSk2TBh6ty7utnAJtJcHgv7JM2ayLQda5oDAIwEoQkYpZdNkYU5SXvT0nj6SHb3MuKNu2nnZeeRlKQsR64/kh1O1sfOBj6R6lPu5QQAwxCaRsH2YQmuOspJ1TyNrkLivjevFtt3qbq8LtO2b9H4nBQbValOTktY0RTelsIeF2PLJlht5sJjJroSLnEMNaV01z6ObnJ5zHtnHpvIVJKx4P2d4STHX3ajJvlovmbg6jkA6JUyZ6W+e3zjLC0tBX9XV1eDv8CNZAPXw2k5it+/yY6bFan5NNMBwKhQ0wRcc62XDfeoo/60KtKu5QIAjAKhCbjm0ku1nua5XKOcrHkCAJwZoQm49tKy+KLTCTwYCEwAMHKEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBIxYfSElqdSErB26ESfQWp0w701JYceNGIXDNZkw87TzTaUKUnejMdxp9mOw/xbOtoXP5RgAMBKEppGoS6FdKLlhak1abipuiJ1CV2HWkrWpa7Cvxxdl3/fFPyiL50adRFSIDzq2bbiYWL2ELWD3x6m2fbjfLiuUXNr2AnBmhKaRyUvNFkzR8GJR0m4KbobWy4Z4k540XroC73BL1k0M6Q4i2Q17DOzL4rgbcQLppf3g+KnMuBGXzQSTseXMK3lsn2U/nsWVOwYAtBGagJNYKEpmYyuo3Wg935Xik6LI3q407bShzWC2dsM29cRrJeOviY/vahKyNSoLa2Gtlp224z4nUcvSVdt5xiaiSP1pVWS+IFn3PC6qgcptijSXx3o/O9gedh1jy9ZVM5SsxereZsl1atfORNt51izbXknG2q8ZUdPjcc2Zien9Xxc27YVDVKN17PZKrG93s+Bxx48xYLkAjA6hCTiROSlk1mXrsCVbL6ZlLl7NdGwzWFNKd0syfWBrMI6kPFmVUruZJiuVoCanJnk3JmGzJLtPfKnNm3nM7krZvm7PLkc4ub5QkUJUE2SnbeZG0vyUfccsjZlXvxAW1YjU5kW8lSP32WbYiEesquRS0TrbZS7Jo2i5umqxjlYakmuHKhsSctJoz7cmGRM0gnWKtvO2WbbJshy59/t+pW+4O7Gh+9Es14OSSLRcdhnMq8oHsc8226s0FU4/WvGk+jBcp+O31zHHwNDjZ/ByARgdQtPI2MIhOrvrPkvETdB8EdQnmSCRkfWfPpLdqbkTN1Plt6PmnrTMLXjteR7LhIPHrrnGW3ncEw6yG/HAkJWCKZjbzYhnMVMJC3EbnFzNRW98Gq6zzsnlsrVY+e3OcqeXyp0gGDR9lqW2FG3hrDy2AeTpST99xOxy7XlSfOCWa6ZgAk5TduO7cb4m+2650w+K4kU1kSMw8PjRLBeAMyM0jUR0huiG7Yw5IyQ43VimQMpsNmTaFlDjnmTc6EvlOqlHg20CGp3O8V2btycHJzm281KI9c2x/YTCQNGSZsOcasx2ljmVyplTD6dpgkai6S0lY8tXIAEE+7sp689dIN2pmGVOruOluKrLBdwwhKbzEJzl4SbK3LMFvg0R8Q7CDWleZkC2fVlmba1NJ7jbJqDzkN2wTUejq8GIL3M4xLZrounNDYmmv8vgyfRkrE9SsN1H1Cx4Jld1uYCbhdB0DuoL5ox5sihz7UIVOG+m0Iw64OwURlzTFBPUYMQ+y/CmPGm6zvF6YfNSdXZAc5898dgrSW7YpfnetHixfl0XYueRlCQZ5k56ldvpttcxRrBcAI5HaBqF7qYRqXHLgRvINicN074S7G5Jmu0+bso+QO1jyDZR2Q6/9rGyGWx8Ucq2g3jwHjM8nDbP3TTj9MtlO2O7eUbDbEPKB/FaNtsXqWaK61hTmvLKPdsxOuj8HZ9/uyO4rc0LO3+3p5kh0bndrHdtRTrrfcL+Vsmmwd6r3Ppur5nHyXV1w0k63Q/cXmc5BkawXACOlzJnJL57fOMsLS0Ff1dXV4O/AHAmNtiYUHoUPymy42ZFaqO6eu80hiyXvaoQwGhQ0wQASvYGp92Ce1lNTkv/20xcjGHLBWB0CE0AoNTTtGaGXKOcrOG5BMOWC8Do0DwHAACgQE0TAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgaodbqhKRSKSnsuBFtLVmbSgXTgmGh7sbj2jlck4loPwbDhKwdummXKjzGeo89AMCoEJpGIiywclKUvBsTV18Yk1KmJr7vm6Em+c2cTKy23FRcH3Up3C1JZtvux2jYl8VxNxkAcKMRmkagtZqT3Se+7C95bkzM4ZqUNj0pf5h1I7LyeMWT5saWiVq4VnYqUjWxuDDjnncJahrjtYhBrVTBRC3LBmtbK2WCV7uWKprm7BRiNViprmA9uLayvmDHmWC+J1Kd7byGYA4Ao0VoGoH00r5UBhSk0tyV5mRR5qLaCFMwji03RfbMeDcK18RMwUSmquS6w45aU0p3SzJ9YGuojqQ8WZVSO9iYMDXbkHIwLRz2l9JuWndtpXlvo1Nbmd2I5ieSj9WCxd8PADg7QtNFifrCzIrUbBOdNKR5JfrCQC8rFRNGavM2OIW1OSftQ5Tfjprz0jK34EnzRTw6N2X9eb/aobpUNvNS24hqK9Oy+CRPbSUAXDBC00XYK8nY3V0pBzUAFckeNk1kyohHX5hrKazZMcN2PmgOG00zmAlkB2WR5THXvBarzQqOl05QC4bZqpsIALgohKbz5k2LZ/4rH5iw5EaFTXZ2PK61mYrU5s3uTNQWncH4ouy7prWjlUZXM2Beam5ae3ixKDTAAcDFITSdt/E5KU42pfRgzTWltGTtYVW8hTkKvGvPNpuZXDwVi7+NptvP4ZV2p41T6XsZ98gIjqGq5IbeqiItnnlL9enpelsBAI5HaBqF9lVPObGNJtEVTGF/l7QsvjiSspRkLHjNmKwvHNFJ9xqK7sPVGXLSWOnsy/RSLbafSzK9XdbXJnZdOZcKOoVHtZPuGGrkEq/pbhbMboS3sxg0HQBwNinf1vPfUEtLS8Hf1dXV4C8AAMBpUdMEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCbgCmmtTkgqlZLCjhvhROODYaHuxl6cQculcZb3AsBVQmgaiboUogItGqbWpOWmmmJD1qZi0y6h0MNo1Bdi+9ENFxEG0kv74vu+HK14bszoBKHmRh2TXd83O/CdAzAChKaRyUvNFGq2YAuGF4uSdlPqC2NSytTctJrkN3MysdqJVLhevJWjzn42Q2XGTRiBKByNcp6jcJbluqx1ym9H+4jvHIDRIDSdt8M1KW16Uv4w60Zk5fGKJ82NrVhNFG6KRDNaakLWDt2EnYKkFtZcDYgZv7MmE/Y17RrJeG1l7H0KwWcmajb7jzudIctl1ylVMK+ISYwbvk52GW2QidfeJWrtzHcn2EaJoevzVDyZnnQPneR+6q0tTNYodpY9XOY1t15mWYL1NY+pyQJeCYSm89bcleZkUebG3XPzIzu23BTZM+PdKNwQppB/JFGNom1KEyk9iAWXzZLsPvGlNt+U0uyulG0NyN66bAUFclYqUa1I8GK99IOieO35WC3Z2miKtzDXru08vSHLNVMw46pSiQWO1suGyHzBvMs6fp2ay2NSmgpr7mzTY/VhtL1asvagJBLV6m3bOZiTj4OKm/cJHG7J+p5I5p7bGsF3MNOpGTbzrs4mg1GuUZYjNz3Yj3c7Ya25vC7TB0dSnqxK7uG0HB2UxdusnCLMAbhuCE0jY35A+5yZtkVnzbNifqxtIdKQZvdrcC3Ygr5fLYSML0plqRNTwjATC8eTZXnsmqi8lccnL/wHMZ9bNkFs/bmLZ0FIyEs5tiznI6w1rT6N4oINaxKrVVWYr8m+W87E9grWwZPiA7cOQUBryu4JzjSqs24f3V2X4kHUPGjC2MNqcvvPVIIgG26/ujwyJzX5J53m9fRSLQhI7XA4X5ZFdxIUfx2Am4/QNBLRGXV05poxZ6axwnSvJGN3bc2CnW7OlA+bJjJlxItqn3CtJPs07bcL0KBAjndAvlu6sNrE7DsmUrgm39bzdZFRhrIhgqAT1bLYoCOxWtWzGPfMNyQWBHcq5rQkL4UT9IsK+zTZGqHYfJx2rVNfnkyPvr89gBuA0HQegrNix5s2P8FdzQpBk50dj5sk6PAvnWYd3zbbuGnnzh5zQRNdWNvTrqE5b+NzUnS1MEFYG0mToBX2Q2rX6s2ayLR9iqY5szSLT0ygXH6UaD5rvIyHqJY0G+5hoLtGyzzfcw8BvNIITeegvpCTatSPKShUmtLp2+KaB0ZWuOBKyXhuv4Z9ci6qpilsKrN9qHKynuk0H52/tMwteCaE1E1Yy4yuSXDnUTKAmuHUV9/NPA6a10rB1XPh8iZClP2sdnNmVgrzEutbZfbkasl8nztNqwBeXYSmUYiuoHFDznYGbt9ywJzpvjgyP/8lGQumj8n6wlG7HwdujuyHtkNwzh0HY7K7cIKapvYxZAK3iVqlu/Zx1MTbafYLLiKIPqPriq2wT1BTMu+comGuvdxuiOY9dLlC6aWyZJZzUspEHcAdxXsHskGn/Z3pDKe7J1ZU25QLPtveAuFopdHpgzjbSNQEZzd8qWU6nx10Go/dQgTAqytlzuB89/jGWVpaCv6urq4Gf4EbzV5sEPSdO00z1hVjA5e9Mi0eVuy44EKKG7B+AK4lapqAGyFsDryoDuDnLbh1QZf606oIfQEBXCJCE3CtRU134V3nb0qzb3CZf1fzXHDvJJrJAFwimucAAAAUqGkCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJqAC9JanZBUKhUOC3U3NlKXQjQtNSFrh240AODKIDSNSKJANMPEastNsVqyNtWZ1ltg4joI9vHUmtmbHf3GDZJe2hff9+VoxXNj4rJSMdN8vyZ5NwYAcLUQmkZhpyBjG0U5Cgo9MxyURZZz7dqC+sKYlDK1cJotFDdzXaEKAABcdYSmEWi9bIhkPEm75zLuScY9lMM1KW16Uv4w60Zk5fGKJ82NLVXtBK6XZI3jaJrZ6gvdNZdWWHtZ2HFPE8178drMcHzndVa/cQCA4xCaRiD9oCherPaovpCT6mRR5sbNk+auNKPHlq2VWm6K7JnxbhRuCBOQH0lUo2ib4URKD3RNd8N4UyZkvxh+tNQXKlKIajpdbWYYirJSmBepPo01Ce9UpCp5Kcy45wAAFULTKIwvyr5/JMWNseAsP2cLzheLnZonyxSoE7YGYFakFvRbaUiTzr7Xz15Jxux+dEMQgCPmOKgsdfZ6EKZHEI7T99r1lrFap6bs7nky7bpHZTcqJh5FwqDUeBnGteyHZRPqK9Kue3paFW/lcez1AAANQtMoBIFoTHaf2LP8Iyk3cqZALbQLqaCgvbsr5aAWwBRuh00TmTLiRbVPuD4my52+a2ZIduru6vB/tzSa2kRvWrxG08y9LhUTtzPtWqfYMbRT6HyuGXKbbrw1PifFyapUgponM4/NvJRj4Q4AoENoGoH6T03hOF+TStDckZbFFyY4mUKqZGsEbIFn/isfxGoCgiY7Ox43SdDhX2Kh6qA8mn1s+8jZGqudijSmHkvBRKe6Dd7RMWRD+2xV8tudMFebD97ppGVuwQub6GzT3HyBWiYAOAVC06gENQHO4Zas74lk7pmz+eAsvxnr29KStYdV8Rbmks13uBnaFwSY/fxgRDVNTvOlSPFBWryphlR+akJU/OIDE5+ipjpb65SoaTLSS2XJb5Zk4mEjdlECAOAkCE0jkN04krLE+rrcLUnGnPUnap7a08dkfeFI9mkeuXHCvkO2aTbcz7sL8ZqmTtNd0A8qel10lVu7eS0nVRO1Snft4+jqOxOIbM3lxnRwQYHtK9XYNMF7ys19fFHK89F7zPBw2jwPJ3XYfk5NM+fYRQkAgBNJ+bYu/4ZaWloK/q6urgZ/gVeZ7URemiKwA8BpUdMEvAqCJjs6gAPAWRCagJssavabbSQvRgAAnBihCbjJZiruirp9WaQvEwCcCaEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBGGynIKlUQeruKQC8yghNI9RanTAFTEoKO25EW0vWplLBtGBY6C6C6lKIpplhYrXlxuMqsfto0L7tHX9xouMuPlzm8iQEoesKLteZnHKfd20LO/BdB64XQtNIhD+iOSlK3o2Jqy+MSSlTE9/3zVCT/GYu9mNp35uTxspROP2gLLI8dkMKl5ulMC9SfdoVeA+3ZH0vL4UZ9/yyTJblKDi+zLCdl+rsFQgoNiTMitQSyzUha4du+isp39ke7rtOcAKuD0LTCLRWc7L7xJf9Jc+NiTlck9KmJ+UPs25EVh6veNLc2DJxydh5JCVT6JaX0sFUGV+Ucr/CGZcu+46JxJuVRFNV6/m6NOcLZq+654lan2SzVn0hrFmIvyYMNmHoThae/cYpzVTkyBxj1Ydr4TFmJJcrHlw6tSZ2+XqnJ0XzOX65zHwfViW/XWlvG7tctfmmrD+37w1rV5PBrnvckBpYG8imzPqZ79dE9JruGtyump2e98emdQfM7tq7aHq4jcxJ0J75jppgGk0/1X4y3/V9EySby486x8mg5QrGdzWT9hsH4HyZM54bq1QqBcPFqfmmWPXz2+6ptZ33ZbLsH7mnwXPzGnPGaV7t+6Zw82XePgoFz+30+HtwRXTv3yO/PBl7HuzbcL9awb6M7cfavN3vZnD7Oz69+7X+Qdn3xPPLB+75ED3vteLLYuaVX+lMTb4+XAe7XJ57TbCc0TEZm0/wvtj6DWe3Ve/yxz878TlWsM7R/MPlipapZ9t3fY96tlcwfdD2M/OKf27XfksuRz9d+12r+3OsxHIPW67uY6/P9gNw7qhpuijRGXHQXFGTvDSkGT+bd2eYYxtFOTooi7e3K003CVdFNtlE19U0V3+arFlJL5Ulv7cuW/H9bJvRNsJXpB8U2/u5+7VhDVZZFsfD5yfmTUu73nN8USpRTaYR/9y2+Zrsu9cENWqNZruWymquTsjYcsYcu7Gao2EOm+YIz4jXtfzpexn3yHzOh+Y4j9Xc2XWWlcfh/O22lbLU2ssd1tAma2A9KR+45Rmfk+JkU3aDlQprubyV2oDtl5WK2weBmULv91GqUumqfToX457ZSpFhy9W9/nWpJGqwAVwEQtNF2CvJ2N1dKQd9GcyPfHeBspmT1MPpsE/Ki0VJN02BNhkr9HBlxJvokk1zLWk2kk02qVTOFL1J3sKctOOLbZ5phxAbyKKmq5ZsbcjZCkR7DLmHQYiYii3X3VJPIM+/Ey+sK+Fx6J7aAFFaNu+INUMeKwgD3UHELMlLs5EyXjjvIOhE4SRc5+ID96l2+e33pr0tzQmFXYa4yaLMtUNRWhZf+FIJAqwJT3vmY+511qBbpynSDl37yTWbtfelbQZ0k0Yu+C3oGLZcQdiNQuZORapnCdUAToXQdN6CM/7YGbEVC0XhmXdearFCKlGw4GoJzv5tQW8L+WYybBj5bdfJtz3sqws2G8iCvm5BLUs8EJycrfUSd4wFFyJIrKO4rckMX6ZkOy93X8BwHE+m2zU/Hc0XTfGmok9Py9yCqz3pt87xzu3REK+JGch+tnvYh+2vlNuMdcgOan672ODoptcyJrydU3AKgrc7gTp2uWIhM6jV7Dr2AJw/QtN5c80GpQfRj65rOohqHFwhnGt3Yq3LI3NGzQ/iVeWa6B7muq6acwFg9gwdc2cem2hTktyDdck8idf0nExY+Jqg/jw2j3YIN8ffg96apuNlpXKiKzv7bI+dQhAK2hc9GEHtSaMpdRMeEutsvxd7ZlucpoN1v8/uFqvJrS/01gjGdUJeJC2eOdc588UaZnvY2rNEZ/mhy5WWxSfm1+LhhJQaZXl82VdsAq8ic0ZzY11YR/B2p9Tk0Om02elsa4dO59ZI2Mmz9324kqL93acTbthZurMvJdZB23bc7d33SSfrbB3q+cy+nY070/MrvR2uBx5z3Z2Xh6x7P8ll679eYQf5ftOS3ws7tJfTLkd35/cu3dulu1N5Z3zZPO90Gj92ewa65zF8vwZ6fie6O6oPX65Q+BrV5wEYuZT9x3xBb6SlpaXg7+rqavAXuOpsLdHYi7KyGQqvHnsbhpJMH+ibfQGMDs1zwFVxuCa55TN2AMeNFjTZ0QEcuDSEJuCyRbejuFuSzDY1COgVXVWXa3RuWQHg4tE8BwAAoEBNEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBIxcXQqplKTcMLHacuOdnUJ7Wio1IWuHbrzRWp0Ixhd23Igr4lTLZddzak261v5kDtdkwm6nhbobAQCXh9A0CtEPezR0FxRd03sLnmMK2XMSFYTxIbFsQwp3K/n+glmLV0FyX/UrzOsLOanO18T3/WDYX0q7KZZ5/2xV8tvhNN/fl8VxN+mSBPvxskLJKIIVAFwQQtMojC/Kvisg7VDLlGSsXQi1ZO3BuhQP3PTtvFRn4wHETJ/KSWPlKJx+UBZZHruwmgYv+lw3VGbcBBv0ZhtSdst9tCJSuhsLRqawG1vOSM29rzZfldyNL/xsYMqJtANPTfKbua6Q25JmQyT/TtY973LYlIbkpRBt5y7ppf3kfrgiLm25ou/WxoDtCQAXiNB0Drwpzz2y0rL4IlabMFMwRWZTdpvu+c4jKe3lpRzVRphCojwvUn16ufU29Z+WpDlfbi93eqlslrsqlSDMmaD3sGoC12OJirLsh2Xx9tZlq6s26iZprZaCGqROcMhKxYTg5sZWLCyafbvnHvbT3DWv6Cdeg9VbqxdI1Px110ieV23lccuV/Nx+r2vGaiTbyxXVvs5WRfbMSUb7vZ1gXl+IxvWuj60ds+Pir0meaBy/XABwUoSmkWvJ1kZzcE1Dl9bLhsh8oR0+bGGQ2zQPGs1LrLXpri0Ja8NM8SaNl3apbDDwpPgganYyBdRdE7LiYfAGar5odgViw5sOw+IvoibYcDtVZ6OCOgwB7aZMGxLMK3JRQd6ukTQBLKq9cmMSbGCK1fzZodPs11VbaeaRGVlt5fDlSjRFHpjgbMblt2MnCSYQ5V6Uw+k2YC4/CkNRVINkxslkWY6Cz7BDpRPEN8JxNXMS0U/TrGNpKlznoxVPqg87NZ3HLhcAnAKhaVTatQDmh1zK8nhAM0bwY24KiZ7p7v1jG0U5sj/ye4NqJEbLFjzhcscL8I7wTH5M1heOgoLJBocOW1jb6WGTlS3cwlB1M9kgmbkX759kjHuSsX//t6iJNgwXnT5LYQiImreCkGBeETVr6pqdopq9Wv9C/3BL1s0xV2uHqKw8tiHi3Gsr61IxAb8drsfnpDjZdQzYQBStY1DL2pDmqGp7TCiKgmP6QTH2nVEsFwCcAqFpVGYqrpA0w5NdGevTMTqsRTIF5otFSRS9mzlJPZwOz7btNNuEMzkdnB0fq6vJ5iSdatsFeTAcSblhliMWnGxtSXQmbwunZE1LU0p3x2T3Sfj+ykxYO9UTKm4Qz6SjnoI36KOUEe9cazDCJr+B29YeL4kmLhO+ly8icnsybcJIO5zZ8JaogbwsV3W5AFx3hKbz0OeM2gamsWWR8kGn+cFK37P1FMkgFTTZZbxksBokHtbs0B3I1NIytxAFonQQEOJn8ra2oxOKwkLJdiLv9O8Jm+ymVUnverKBMVnTZpwk4J5auL2HSjRxueHcO0+748SGfhvW7pZEBtWGXairulwArjtC0zkIOgxPFmXO/Uh3AlOfPhVBwKpKrl3DU5dHy/o+USNzuCa52Odm38kHhU67X4zrsB5e9RUGrOZyrt2xtnudb6KgCSi+Tcy+Cm4f8OS0QVUr3N7V2QG3dbDHkO07NLLO30rmmCnZmtNYUEveXkEh6hM2yg7ao1guAOjH/KDcWKVSKRjO3Xbet5uyPUyWfXPW79R8Ez+S0495TX7bjT5XR355MrY84vnlAzcpklivvFnKpKMVrzM9sT432EHZ99rbpN++CvflwH0YbNPebdlzDAVDcp8ktrcZvJX4Fu89zrTHUfd8g2HeLeExy1Wb754WWy773p7jvPc4S35+tG36f2+ieQfviZbRCvZLZ7sOXS4AOKWU/cf8oNxIS0tLwd/V1dXgL4ARsrcNuLsr5dgVb+E4e1+yS7xS7aouF4Brj+Y5AKfT555TrefrZtx5d4w/xlVdLgDXHqEJwOnMVMI7wSeu2rN3iU9e7HDhrupyAbj2CE0ATi26AWVnuBrB5KouF4DrjdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQB10ZdCqmUpKbWpOXGjF5L1qZSUthxTwEAbYSmEasvmELNFGzJQicsiOz4YFiou/ERVxi6YWL1/IpEnNHhmkzE9lUqNSFrh27aTbDzSEpSlscz7rnRWp2Ira8ZzjW0ncBO4eosC4BXAqFplMyPeK6Rl/yke+7UF8aklKmJ7/tmqEl+MxcLRjZQ5aSxchROPyiLLI9xpn8lmXB7tySZbbsfo2FfFsfd5HOXlYr9zBeLknZjRq3+tCr5J7H5m2N6bDkjtfb6nu/nA8BVRmgaGVOgztoC57FMuzGBwzUpbXpS/jDrRmTl8YonzY2t8AzZntnv5aW85Iqh8UUpz4tUn3bXRuHS7VTE7GEpxGph4oIamXgtYlArVTBHhmXDsa2VitcqRtOM9mtj02O1KInanp6aSitZW9ldA5OsLYp9blxwrCbXz4YomS+Yo7af3qa8xDawNUHmcfyzu2tRk8vVXUObXKf2e6PaPvN9k72SjLVfM2C9AGBECE0jUl/ISXW+JpXuArW5K83JosxFtRHBmXvT/Nib8eZp62UjUSjZQiS3aR40molCD1fATMFEpqrkTl04N6V0tyTTB7bG5kjKk1UpJUKEnXc0vSZ5EwgeuRCRXtoPanmOTODuxx5/7drK7tqgrtqio5WG5Po0a9V/WhJZeZwISNl38iKbuQFBTcG8d+xFOVym7bw0lx91tp0JP7muWqzO96erBtZsj0xUA2tOLPbd/GSyLEft91cGhDsAGA1C0yi4M/TaxpCf7PbZsZhCwhSI0pBmvC+MPSs308c2inJ0UBbPhSpcJWHzWG3ehpuwduOkzaj57ag5Ly1zC540XyT3cmd6VgrzJju/1Efndu1ll6DJbbsTKNJLZRPI1mUr0RerLpVNT4oPuhreZipBYLFNyqeqzbGhJvpeBKGz67g3QbHSbxsebsm6lKUW1cCapbc1tNTAArhMhKYRWHtg+7kMOcu1TQh3d6UcnQ0fNk3RkREvqn2yBdLD6fCM2dYQBLVT09K/TgGXLbvhaja281KdHWXH/WTTmP2c/XZoGC67cWQiRqepqhPmWtJsmGhiljMMPXbImaiS1FotSXW+PKB/lutL1Q6MI+r8bmuM3DYMlite+2W/A4mmN3NCYWtoAeASEZpGYH0vXiiNSSl6bgsBz4YfT8oHsVAVC0Xpexnzb15qseaUoMku49HZ9qqbqZgQYXZnV23R5UjL4gsX5g7K0jDHX7wWLJ/ovG6HeAf2lmxtNCX/zpCaUie7YWtJm7I7qlUOarLCZaplTEiKB6dE05sbhtXmAsA5IzSNQNC/oj3YviqukLJBaHxOipNNKT2ICoOWrD2sircwF4aiqJ9Mu89IXR6ZM2pNAYbLZpu0RLypWJ1guy9aPbjS7lLi1LgnNoqHwmbA6uyQZrU+txkYKOgM78l0fJWjJsSov94pJbaj/V7slSQ3rBbPnpD0NDMCwPkhNJ07WwMQbzoZk/WFo1izi236iPcZyYmYwNXToRyXrvtKL7uvbEflaF+ml2qx/VyS6e3yiJpYw6vU7GcGoSQ6VmJBO3HlnFuu6BiynciDzt/x18RqdGyfp3aIT+h8bnuYbUj5oNMva/F5eIuMYNrDaakN6KjeT/f2DDqrt2tcw++F7fwdf02iD9n4ovk8kdLdaDpXzwE4XynfVo/cUEtLS8Hf1dXV4C+ALvYChaC/HVeeAcBxqGkCXmXB5fsEJgDQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0DRCrdUJSaVSUthxI9TqUjDvs++1w8Rqy40PRfMNh4J5ddzw92LUkts7tZDcG9dTS9amTnHc7hQ628ENHH8AbjJC00iEhU5OipJ3Y/Tse3PSWDkS3/fFPyiLLI91CjBTMI0tZ6Rmp5mhNl+V3NSaeZd1zHsxYjYw5US2w33h+zXJb+Ze8aCQbx+b0fFHcAJwUxGaRqC1mpPdJ77sL3luzAnsPJLSXl7KS+nw+fiilOdFqk9tDYYJRQ+r4q08lmw4VbIflsXbW5etQ/Nk6Hsxaq3VklTna1KZcSPMXqls56W5sWX2VFgDlQys3ePCcN2umYnXUh2uyURQixiryWqHY6erZicZTobXgCVrKzvLVF+wz8fMcWSOm9nO9FMFH3P87dvtsfzILI3TtcydbaHZXgBwtRCaRiC9tB8rSE+m9bIhMl9ohyJbuOU2zYNG0xSYTdnd86T4wIUiW6jcLZmxZnzzuPdi1JovmuJNdQVjb9qF2KwUugPrYVMakpeCOzbqCyacZGqulupIyo3uWqqq5FIlmT5wtVh7JXkUBQgbPmYbUg6mhcN+FJaN+kJFClGNj6sBa4cPE8hysdpKO0THa3bDPjfLMimSb9egJed9InZ7mLVu2lBvj9enhfY8fROoqrNR83Kf7bVTMVugs70A4KohNF0V7ox8bKMoRwe2NmnXRKNIVEMRNg3VTGHTeBkrbIe+F6PSNBk1c68rTIx7knEPg1rAzYoLBWavPV8XadcS1qWymZfaRhRx07L4JKql6shv78viuH0UhopwP0c1jjU3rVd2o9IOz8n3RqpSuYganNj2sMtRaa+vMVMwkSgKVL3bq/40WasKAFcNoekq2MxJ6uG0HNmz8ReLkm6a0DNpz9itppTujgXNf2ENQStZeA99L0bJM2kgGUSMoDYpI54NM+NzUpyMwklLtjakU0sYvM7WJHWaqlKz1XBaW7KWxdYChTU+tsaxT2CL62oGC2ocI67ZrN381t3sN0rBenaEzX/RkDNbICaxvcJQ2W5qBoAriNB0ydL37Hl5Xmo28ISjwma3jGeeezI9aQrrlaNY81/YZDdtUtHw92LUbNOcbaJLSITUtMwteGGT0+GWrEtR5hI1Q7FO09EQ23eDhcfBQLY/lAlg8eY1WxuZMFPpTMuUZOycgpOtXWu6EBk2F8fXuWa2QFxse9mmuVhTMwBcRYSmixJ09O1zlh80WVQl1+64W5dHy03Jv2OLj7BQaS7nZM01aQSdkSddYTz0vRi19IOiePG+QmZ7F2xYedIJPsFrGk2pm/CQiY2PalU6++okXLho9wfqJwzSgZ1CsqapS0+/LDN/W4t25gsIzOeO2eNvO9ZUGKv1rC901TQZ6aWy5DdLMvGwIeUPOW4BXHHmDPDGKpVKwXDutvO+3ZTdgznzjznyy5N2fN43Z95dar45Ax/wPvPOFa89TSbLZk5xw9+LETso+yYEDN3etXk7rd9+jo6BzuCtuL0ZzLffezoSx0H8vUb4mW4wx0jZPI+md79PcwzG5z1Qz3Hv+eUDNy3QPc+yed79GrfsPcc1AFw9KfuP+UG7kZaWloK/q6urwd/LZvt35BplOVI1yQCvBvu9KE0dnf6KPQC4IDTPXQTXSZfABHQJmhLpAA7geiA0XYSoEy6BCQhFV/sF956K3y4BAK4uQhOAi9e+mi+6LxUAXH2EJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCbggrdUJSaVS4bBQd2MjdSlE01ITsnboRl86zXK510ytScuN0Qi2R892AICri9A0QlGhWNhxI9pasjYVFTxmOGFBkShsUwVTRMXFC7WUTKx2FVs7hdh7T1YY1xc68+0pEA/XZKI9XzecYL2Gr9Mx2+sM63RWiW3iht793V96aV9835ejFc+NictKxUzz/Zrk3Rit5LY82TId7/TLdXbhcTC6dQGAsyE0jUT4456TYt+Cpb4wJqVMLSgwg8JnM9cbbgYxAWFsOSO14L2+1OarkmsHGPu5OWmsHIXzPiiLLI91ChkbbGYbUj4I33u0IlK62x1QBjCfm5NomY+kLCUZ6w4vk2U5cssVDBtZN+EYQ9fpmO11lnUaES/a3m6ozLgJlym+L7bzUp29yLDhgtWLRUm7MQBwExGaRqC1mpPdJ77sL/WpQTCFfGnTk/KHUaDIyuMVT5obW+2QMJgJRQ+rppB+bN4Vyn5YFm9vXbZs7crOIynt5aW85Iqq8UUpz4tUn4YRov7TkjTny7I4HjyV9FLZhLqqVDSF6UwlFoLSMrdg1q3RVCzzcY5Zp2O215nW6Zz1NDcFtXEXG+gCZt/Z2qzqw04QTdZGdWrngvFdtYj9xg2SmG+fmsbk5/Z/Xbz2Lgp64TgTnvfM8WwCYDRdfbIBAOeA0DQCttllYG1Dc1eak0WZc4V8WMvSFNkz492owZqyu+dJ8UF0/l6Xwl0TGux48+bWy4bIfKEdPmwBlds0D4Jw05KmmZx/pz01qJWqmkeNl5dZ8Axfp+Hb66qu09WTvpfpHGMmvD1q1xq62rkHYShKPyh2AmugJVsbTfEW5lS1RkObHM3n5palXStYM4E+qBGL10hu5qQ0FdbcxYNedsO+50jKk2Z/b4fvt8N+dIIAAJeA0HRRgloHc7Y8K1IL+oc0pKnui2PDgT3TzomYAsQWPomQYIKFPQsf2yjK0YGttUkGsuisfX3hKCiYmi/iUxWCwq8p+SddzS97JRlzNQC9/ZKOc8w6HbO9zrxOZ9BcHnPrbIer1Gk7xpuWdowZX5RKLGyEQckdI0HtZFPWn7ttf7gl6/HayzNoPV9PBODsO/nek4X5WjsIJZYLAK4gQtNFsOHi7q6Ug7PlimQPmyYCZMSLalOGakrp7ljQ/GfPtCszYW1L5p4r1MyZeurhdNifxfYpCWpqOgWmbdqIzuRt4WTDhTflpkbBpD30Cz5hTZCsHCVr00xhux+sTzgcrTQkpw5Ox6zTMdtr6DpdgGSfpv12U+GVYo8D97ATUN0Q1Ox12DATNX/aoCOxptOzCGu7OrVY9afVRM0oAFw3hKbzFpzxe1I+MIW/GxU2QcVqAgbyZHoyLKQ7gSVs3po2bw4KJclLLdYBN2iyy3jmeVo8Ozl2Jm8Lz0Q46Qo+QUAJpzj2yrycVBPz6C+oJXCPhxu+TsO3l2KdEAgCijvGgo71Eusobmsjw5eFZgqSD8KNbZqTWNPpGQX70gbkMKzlNs2xqr1YAACuIELTeRufk+KkKThcH5LgrN92hO7uM+Ka2JKdZMMO2M3lXKzjbkmqUZOHLeykKrn2e+ryyDajuT4/QXPIZq5zFZXrOF5QXe3VCUzHXxVn1umB7aDdVYtwmnU6ZnudbZ0uQLuzfNRX6+KFfdtM8Hwea04NgrTl9lXwOGI729t+TjlZz3Q62Z+V7bRvaygHh/LjhCE5urABAC6d+TG7sUqlUjCcu+28bzdl95DfdtP9I7882RnvrZhz/h4138QBXybL5tVJRyteZ74909373ND5TCexbHnzap3EZ8aGaP7d00e7Tsdsr1Ou0yjU5getqxVfbs8vb5d9r718yXVqD/Nu6fseQ2YeB+HkYXr3Vdc2ObDL0ZmeX4kvl+NeM/z4iYZouY5Zp67PDYfO5wbLHb3WCl7fvT+Tx/fgbQ8A5y9l/zE/RjfS0tJS8Hd1dTX4e7XZfifu/kQ3pgnjJq7TDWX7twX9yE5aGzSY7axv+57Fm3btuOD+XxwPAK4hmueugPBeNjcrXNzEdbq5wia7UXUAD4V9zZLqUtmUC+20DwCjRE0T8MpyNYF75qGq79oJBbVXyf5T3kqy5gkArhNCEwAAgALNcwAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAAAUCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEzBi9YWUpFITsnboRryy6lJI2W0xfHucZnu1VicktVB3zwDgYhCaRuFwTSbahYMZptak5SYFuqYXdtx4paCAaL+/YIqiuHjBlJKJ1cQni+wUYu89eUEefXbPMifma4aeAuyY5RqqJWtTQ+Z9zDoN316nl5xvOJx0X56fcJudanm6j9+RBb6sVHxffL8meTfmItkwdrLjDgCGIzSNwvii7AeFQzjUMiUZaxf0pjB7sC7FAzd9Oy/V2RMUSiYgjC1npBbNe74quXYoswVlThorR+G8D8oiy2OdgtMWhrMNKbvPPloRKd3VhoiwEM5JsX+BN1MJPzMYjqTcyMUKqGOW6xj1hTEpZWpu3qbA3YzN+7h1Grq9RmCyLEdu3uG+7A0q2Q07fV8Wx92IK82E27slyWy7dQqGi13267W9ALzSzI/kjVUqlYLhoh2teL7Mm2K7r5pvQoif33ZPhzryy5PieyummI4clH1PPN+EBt8U2r5I3syxozYv7c+OPw7pP9uuQ/g63XvsZ7WX85jlGiq+fk6wPSfLZmsct07HbK8zii9HJDEu+CyzfMGQXH8r2kbBe9zrEts12G7R+/ssc2J6Zz2DbRIb3z19qD77KsFOT6yz3d7RstntbR+H+yD83H7zir8n5pjtlZyefF2wDc1xEF/3aFvGt29i0Bx/ADAENU0j15Ktjabk38m652fRlN09T4oP0u55WCvQtOOb5pNeNkTmCxJ9km0+ym2aB42mWYqWNM3kznKEtT9V86jx8vh6l/TSvlRm3JNj1aWy2VnO4ct1jOauNCeLMhfVOgQ1R2Zl98z4Y9dp+PY6D+l7GbdsRlTjeFAWU9j31Vwek7EX5aBGxxTuUn3oasG6atD87UxPDVoqPt0M+0vheoY1NUdiAqOY4NAzfaiZguTNFsyduhmzaZazJNPBctllqEpJ2yQ2dHuZffugJBLVVm6bWGZeVT6otI8r2cxJaSqcHt+W9ti140ygEhMcw/fbYaP9TgA4FULTqLT72YxJScryeEDgqC+YQn5y8PT+wqayVConYgpFWxgkgo/77LGNohzZAigqxJ2wo+2YrC8cBYVL88VoEkSnj49Zp/lyb/PKMcs1VNTPZlakFvSJaUgz1qQ5fJ2O2V6j5E0PDEh92eY9V3inHxTb26T1fN0EhFpnG848DgJIJWj6M+vzsGoCQGz6yIT9joJmzGBfnrxfVH47alpLy9zCiI6vwy1ZjwfgINx1hd/5WjsYxrclAJwXQtOoxPv4PNmVsT5n7mGNS15qLxZN8aJlz+THZPdJOO/KTFjbkrnn5mDOtlMPp8N+Nna+QU1NpyC3fW6is3FbwNgCzZtyU3s6AJ+stiE6o7fD0VQp2QH+mOUaaq8kY3d3pRzMuyLZw6aJTBnxXGAYuk7Hba9Rs+vlHmp4C3OdfR/UtIQ1J3YdbC1UZ1+Y8L0Xvsyu0655fG7rYIS1VWZw/bQuvQP1uGf2eFPWn7vl2KlI1cSmwolONgBgtAhN5yE4K07WjNjANLYsyeaFY3kyPRk2MXSaysImqGmTEYKmIfNJ8RAWNI1lPPM8LZ6dHDsbtzUWiQDR1YE9CCjhlBOLn+kPX65jBDU3Xc0w7cB13DoN317nof60KqINg8dINCW5IVyPcL0uhAn/tmZuJLVFZxKucztIzprItH364xMARoHQdA5aqyWpxvrldALTkCuEoua9xOX1rrljOde+2i4x76g/Svs9dXm03OlPlX0nH9T4tJtbdh5Jae98ztbrPy1JM+rHdMxyhaImtK4rCcfnpDjZlNKD2BWCtmnK1dAMX6djtteIhTWHJuA9P0nNYX92veLLnRSuV3V2WE1gGCirT09SV9iP7Z9mIku75s5oN3vZfRb2ITt3dr9K7ErFdoDUs+vQ3NhyxxEAjID5MbqxLuzqua6rmnqvNopN6/say72uZ3zX1UCD3ueGnqvcEss25Cqpbt3r5IaBVyj1XJl0zHIZ0Tx6p4VXwUXv7bkK7Jh1Gr69Tq9nnbs+u3d68jX2Sq+edYnr2ebD5987r+Q2H/pZTr9l7n6fXe5oWn7bfkb86rnk/gvmFx0LfY+hzlV0w7dX8hiIhsTxFz/mgivtuo+Frnlw9RyAM0rZf8wPyo20tLQU/F1dXQ3+Xm32LN7dn+hVucqnfUUY9+hBF3ts2D5x8f5/wfFiLwygmQ7A5aB57goIr0J7lQKTu1s4gQkDBH3guoyy/xgAnAY1TQCuIFfz2r6C0LC3a4jXPAHABaOmCcAVlJbFF8krCYNbV7ipAHAZCE0AAAAKhCYAAAAFQhMAAIACoQkAAECB0AQAAKBAaAIAAFAgNAEAACgQmgAAABQITQAAAAqEJgAAAAVCEwAAgAKhCQAAQIHQBAAAoEBoAgAAUCA0AQAAKBCaAAAAFAhNAAAACoQmAAAABUITAACAAqEJAABAgdAEAACgQGgCAABQIDQBAAAoEJoAAACOJfL/A1fVIn2DdZO6AAAAAElFTkSuQmCC"}}},{"cell_type":"raw","source":"Obtener únicamente datos donde el reloj se usa","metadata":{}},{"cell_type":"code","source":"def acotarDatos(actigraphy_df):\n    #Solo datos donde el reloj de actigrafía se use\n    actigraphy_df = actigraphy_df.loc[actigraphy['non-wear_flag'] == 0]\n\n    #Datos donde la luz pertenezca a un nivel de oficina\n    actigraphy_df = actigraphy_df[(actigraphy.light >= 320) & (actigraphy.light <= 500)]\n\n    return actigraphy_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:15.208611Z","iopub.execute_input":"2024-11-27T17:03:15.208923Z","iopub.status.idle":"2024-11-27T17:03:15.214050Z","shell.execute_reply.started":"2024-11-27T17:03:15.208891Z","shell.execute_reply":"2024-11-27T17:03:15.213030Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Prueba para un participante","metadata":{}},{"cell_type":"code","source":"actigraphy = pd.read_parquet(f'/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=0417c91e/part-0.parquet')\nactigraphy.drop(\"step\", axis=1, inplace=True)\nactigraphy = acotarDatos(actigraphy)\nactigraphy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:15.215523Z","iopub.execute_input":"2024-11-27T17:03:15.216192Z","iopub.status.idle":"2024-11-27T17:03:15.311229Z","shell.execute_reply.started":"2024-11-27T17:03:15.216145Z","shell.execute_reply":"2024-11-27T17:03:15.310218Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**GENERACION DE VARIABLES ESTADÍSTICAS GENERALES**","metadata":{}},{"cell_type":"code","source":"sensorFeature = actigraphy.columns.to_list()\n#for axis in ['X', 'Y', 'Z']:\n#    if axis in sensorFeature:\n#        sensorFeature.remove(axis)\nsensorFeature , len(sensorFeature)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:15.315693Z","iopub.execute_input":"2024-11-27T17:03:15.316020Z","iopub.status.idle":"2024-11-27T17:03:15.322498Z","shell.execute_reply.started":"2024-11-27T17:03:15.315988Z","shell.execute_reply":"2024-11-27T17:03:15.321406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get Statistic Feature list\nstatFeat = actigraphy.describe().index.to_list() \nstatFeat, len(statFeat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:15.323887Z","iopub.execute_input":"2024-11-27T17:03:15.324154Z","iopub.status.idle":"2024-11-27T17:03:15.357744Z","shell.execute_reply.started":"2024-11-27T17:03:15.324126Z","shell.execute_reply":"2024-11-27T17:03:15.356769Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Combinar para cada variable de actigrafía la estadística general**","metadata":{}},{"cell_type":"code","source":"# generate senor feature statistic column list\nsensorStatCol = []\nfor sta in statFeat: # start statist row \n    # print(sta)\n    for item in sensorFeature:\n        # print(item)\n        sensorStatCol.append(f\"{item}_{sta}\")\n        \nsensorStatCol, len(sensorStatCol)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:15.358721Z","iopub.execute_input":"2024-11-27T17:03:15.358984Z","iopub.status.idle":"2024-11-27T17:03:15.366484Z","shell.execute_reply.started":"2024-11-27T17:03:15.358958Z","shell.execute_reply":"2024-11-27T17:03:15.365575Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Leer archivos csv y paquete de datos parquet**","metadata":{}},{"cell_type":"code","source":"def obtenerPromedioUsoComputador(actigraphy):\n    # Datos donde el reloj se estaba usando\n    actigraphy = actigraphy.loc[actigraphy['non-wear_flag'] == 0]\n\n    # Filtrar filas donde la columna 'light' esté entre 320 y 500\n    actigraphy = actigraphy.loc[(actigraphy['light'] >= 320) & (actigraphy['light'] <= 500)]\n\n    actigraphy = actigraphy.loc[actigraphy['enmo'] > 0]\n\n    # Filtrar por ángulo de la muñeca para uso del computador\n    actigraphy = actigraphy.loc[(actigraphy['anglez'] >= -30) & (actigraphy['anglez'] <= 30)]\n    \n    # Validar si el dataset tiene datos después del filtrado\n    if actigraphy.empty:\n        return float('nan')\n\n    # Día en el que se realizó el registro\n    day = actigraphy['relative_date_PCIAT'] + actigraphy['time_of_day'] / 86400e9\n    # Diferencia de tiempo entre registros consecutivos en segundos\n    interval_seconds = day.diff() * 86400\n    time = {\n        'step': actigraphy['step'],\n        'day': day.apply(lambda d: round(d)),\n        'elapsed_seconds_bt_register': interval_seconds\n    }\n\n    time_df = pd.DataFrame(time)\n\n    # Combinar con el DataFrame original\n    actigraphy = pd.merge(actigraphy, time_df, on='step')\n\n    # Explicitly copy the DataFrame to avoid the SettingWithCopyWarning\n    data = actigraphy[['step', 'day', 'elapsed_seconds_bt_register']].copy()\n\n    # Define a threshold to break continuity (10 minutes or 600 seconds)\n    time_threshold = 600\n\n    # Safely set 'is_continuous' and 'interval_group' on the copied DataFrame\n    data['is_continuous'] = data['elapsed_seconds_bt_register'] <= time_threshold\n    data['interval_group'] = (~data['is_continuous']).cumsum()\n\n    # Group by 'day' and 'interval_group' to get total computer usage time in each continuous interval\n    grouped_intervals = data.groupby(['day', 'interval_group'])['elapsed_seconds_bt_register'].sum().reset_index()\n\n    # Calculate total computer usage per day\n    daily_usage = grouped_intervals.groupby('day')['elapsed_seconds_bt_register'].sum().reset_index()\n    daily_usage.rename(columns={'elapsed_seconds_bt_register': 'total_seconds'}, inplace=True)\n\n    # Calculate the average computer usage in minutes per day\n    daily_usage['average_computer_use_minutes'] = daily_usage['total_seconds'] / 60\n\n    # Calcular el promedio de uso de computadora por día\n    average_daily_use = daily_usage['average_computer_use_minutes'].mean()\n    return average_daily_use","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:15.367622Z","iopub.execute_input":"2024-11-27T17:03:15.367877Z","iopub.status.idle":"2024-11-27T17:03:15.380267Z","shell.execute_reply.started":"2024-11-27T17:03:15.367851Z","shell.execute_reply":"2024-11-27T17:03:15.379461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def loadParquetFile(directory, fileName):\n    \"\"\"\n    read parquet file \n    \"\"\"\n    path = os.path.join(directory, fileName, \"part-0.parquet\")\n    #print(path)\n    df = pd.read_parquet(path)\n    computer_use_mean = obtenerPromedioUsoComputador(df)\n    df.drop(\"step\", axis=1, inplace=True) # drop step column\n    statDF = df.describe().values.reshape(-1)\n    \n    \n    return statDF, fileName.split(\"=\")[1], computer_use_mean # get ids\n\ndef loadTimeSeriesData(directory):\n    \"\"\"\n    input : root direction\n    \"\"\"\n    print(\"DIR :\", directory)\n    filesIds = os.listdir(directory) # get list of folder name (files ids)\n\n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: loadParquetFile(directory, fname), filesIds), total=len(filesIds)))\n#     print(results)\n    statistic, ids, computer_use = zip(*results) # pack into Statistic and Ids \n    \n    # create new dataframe with n statistic sensor data\n    data = pd.DataFrame(statistic, columns=sensorStatCol)\n    data[\"id\"] = ids # add ids into dataframe\n    data['computer_use_mean_min'] = computer_use\n    \n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:15.381423Z","iopub.execute_input":"2024-11-27T17:03:15.381716Z","iopub.status.idle":"2024-11-27T17:03:15.396884Z","shell.execute_reply.started":"2024-11-27T17:03:15.381687Z","shell.execute_reply":"2024-11-27T17:03:15.395988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\nsample = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')\n\ntrain_ts = loadTimeSeriesData(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\")\ntest_ts = loadTimeSeriesData(\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:03:15.397972Z","iopub.execute_input":"2024-11-27T17:03:15.398325Z","iopub.status.idle":"2024-11-27T17:04:53.268712Z","shell.execute_reply.started":"2024-11-27T17:03:15.398293Z","shell.execute_reply":"2024-11-27T17:04:53.267735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ts.sort_values(by = 'id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:04:53.270089Z","iopub.execute_input":"2024-11-27T17:04:53.271017Z","iopub.status.idle":"2024-11-27T17:04:53.309492Z","shell.execute_reply.started":"2024-11-27T17:04:53.270969Z","shell.execute_reply":"2024-11-27T17:04:53.308400Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.sort_values(by = 'id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:04:53.311230Z","iopub.execute_input":"2024-11-27T17:04:53.311576Z","iopub.status.idle":"2024-11-27T17:04:53.344662Z","shell.execute_reply.started":"2024-11-27T17:04:53.311540Z","shell.execute_reply":"2024-11-27T17:04:53.343488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = train_ts.loc[train_ts['id'] == '7b8842c3']\n\n# 1. Obtener las columnas categóricas\ncolumnas_categoricas = df.select_dtypes(include=['object', 'category']).columns\nprint(\"Columnas categóricas:\")\nprint(columnas_categoricas)\n\n# 2. Obtener las columnas con valores nulos\ncolumnas_nulas = df.columns[df.isnull().any()].sort_values()\nprint(\"\\nColumnas con valores nulos:\")\nprint(columnas_nulas)\n\n# También puedes mostrar cuántos valores nulos hay en cada columna si lo deseas\nprint(\"\\nNúmero de valores nulos por columna:\")\nprint(df.isnull().sum())\n\nplt.figure(figsize=(8, 5))\ndf.isnull().sum().plot(kind='bar', color='skyblue')\nplt.title('Número de valores nulos por columna')\nplt.ylabel('Número de valores nulos')\nplt.xlabel('Columnas')\nplt.xticks(rotation=0)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:04:53.345829Z","iopub.execute_input":"2024-11-27T17:04:53.346129Z","iopub.status.idle":"2024-11-27T17:04:54.237732Z","shell.execute_reply.started":"2024-11-27T17:04:53.346099Z","shell.execute_reply":"2024-11-27T17:04:54.236633Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**GENERACIÓN DE TENDENCIA DE ARCHIVOS PARQUET PARA CADA PACIENTE**","metadata":{}},{"cell_type":"code","source":"def generate_trend(id):\n    actigraphy = pd.read_parquet(f'/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id={id}/part-0.parquet')\n\n    #Solo datos donde el reloj de actigrafía se use\n    actigraphy = actigraphy.loc[actigraphy['non-wear_flag'] == 0]\n\n    #Datos donde la luz pertenezca a un nivel de oficina\n    actigraphy = actigraphy[(actigraphy.light >= 320) & (actigraphy.light <= 500)]\n\n    serial_data = actigraphy['enmo']\n    print(serial_data)\n    \n    if(len(serial_data) < 10):  \n        print('SIN REGISTROS DE LUZ DE OFICINA')\n        return\n\n    # Aplicar rbeast a la serie temporal\n    result = rb.beast(\n        serial_data,\n        season=\"none\",           \n        trend=\"linear\",         # Probar una tendencia cuadrática\n        #changepointmax=3,        # Permitir más puntos de cambio\n        #variance=True              # Considerar variabilidad en los datos\n        print_param=0,\n        print_warning=0,\n        print_progresss=0,\n    )\n\n\n    # Aumentar el tamaño de la figura\n    plt.figure(figsize=(20, 12))  # Ajusta el tamaño según tus necesidades\n\n    # Visualizar los resultados\n    rb.plot(result)\n\n    # Mostrar el gráfico\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:04:54.239192Z","iopub.execute_input":"2024-11-27T17:04:54.239643Z","iopub.status.idle":"2024-11-27T17:04:54.247641Z","shell.execute_reply.started":"2024-11-27T17:04:54.239595Z","shell.execute_reply":"2024-11-27T17:04:54.246437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"generate_trend('79bd36b6')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:04:54.248842Z","iopub.execute_input":"2024-11-27T17:04:54.249159Z","iopub.status.idle":"2024-11-27T17:04:54.307454Z","shell.execute_reply.started":"2024-11-27T17:04:54.249128Z","shell.execute_reply":"2024-11-27T17:04:54.306367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for id in train_ts['id'].iloc[0:20]:\n    print(id)\n    generate_trend(id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:04:54.308610Z","iopub.execute_input":"2024-11-27T17:04:54.308913Z","iopub.status.idle":"2024-11-27T17:05:22.723795Z","shell.execute_reply.started":"2024-11-27T17:04:54.308884Z","shell.execute_reply":"2024-11-27T17:05:22.722676Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### COMBINACIÓN PARQUET EN UN SOLO DATASET","metadata":{}},{"cell_type":"markdown","source":"**AHORA SE PROCEDE A AVANZAR CON LA COMBINACIÓN DE DATOS DE CADA PARTICIPANTE A NUESTRO DATASET PRINCIPAL**","metadata":{}},{"cell_type":"code","source":"df_process = pd.merge(train, train_ts, on='id', how='inner')\ndf_process","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:22.725023Z","iopub.execute_input":"2024-11-27T17:05:22.725336Z","iopub.status.idle":"2024-11-27T17:05:22.765183Z","shell.execute_reply.started":"2024-11-27T17:05:22.725307Z","shell.execute_reply":"2024-11-27T17:05:22.764164Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Eliminar algunas colunmas que podemos reducir","metadata":{}},{"cell_type":"code","source":"#ELIMINAR RESULTADOS DE CADA PREGUNTA DEL CUESTIONARIO\ncolumns = [f'PCIAT-PCIAT_{i:02}' for i in range(1, 21)]\ndf_process = df_process.drop(columns=columns, axis=1, errors=\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:22.766635Z","iopub.execute_input":"2024-11-27T17:05:22.766956Z","iopub.status.idle":"2024-11-27T17:05:22.773545Z","shell.execute_reply.started":"2024-11-27T17:05:22.766922Z","shell.execute_reply":"2024-11-27T17:05:22.772486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#ELIMINAR TEMPORADAS\nseasonColumns = ['Physical-Season', 'Basic_Demos-Enroll_Season','CGAS-Season0', 'CGAS-Season','Fitness_Endurance-Season','FGC-Season','BIA-Season','PAQ_C-Season','PCIAT-Season','PAQ_A-Season','SDS-Season','PreInt_EduHx-Season']\ndf_process = df_process.drop(columns=seasonColumns, axis=1, errors=\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:22.774929Z","iopub.execute_input":"2024-11-27T17:05:22.775318Z","iopub.status.idle":"2024-11-27T17:05:22.788252Z","shell.execute_reply.started":"2024-11-27T17:05:22.775284Z","shell.execute_reply":"2024-11-27T17:05:22.786647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# COMBINAR TOTALES Y ZONAS DE SALUD EN UN INDICADOR\ndf_process['FGC-CurlUp_Status'] = df_process['FGC-FGC_CU_Zone'].apply(lambda x: 'Healthy' if x == 1 else 'Needs Improvement')\ndf_process['FGC-PushUp_Status'] = df_process['FGC-FGC_PU_Zone'].apply(lambda x: 'Healthy' if x == 1 else 'Needs Improvement')\n\n# OBTENER MEDIA\ndf_process['Average_Grip_Strength'] = df_process[['FGC-FGC_GSND', 'FGC-FGC_GSD']].mean(axis=1)\ndf_process['Average_Sit_Reach'] = df_process[['FGC-FGC_SRL', 'FGC-FGC_SRR']].mean(axis=1)\n\n# SELECCIONAR COLUMNAS A ELIMINAR\nFitnessGramChildColumns = [\n    'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone',\n    'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone',\n    'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone',\n    'FGC-FGC_TL', 'FGC-FGC_TL_Zone'\n]\n\n#ELIMINAR COLUMNAS\ndf_process = df_process.drop(columns=FitnessGramChildColumns, axis=1, errors=\"ignore\")\ndf_process = df_process.drop(columns=['Physical-Waist_Circumference'], axis=1, errors=\"ignore\")\ndf_process","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:22.789955Z","iopub.execute_input":"2024-11-27T17:05:22.790336Z","iopub.status.idle":"2024-11-27T17:05:22.838330Z","shell.execute_reply.started":"2024-11-27T17:05:22.790302Z","shell.execute_reply":"2024-11-27T17:05:22.837289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#PROCESAR INFORMACIÓN Bio-electric Impedance Analysis\n# cREAR COMBINACIONES DE COMPOSICION DE AGUA, MASA CORPORAL, MASA MUSCULAR PARA UN ANÁLISIS REDUCIDO \ndf_process['Total_Water_Composition'] = df_process[['BIA-BIA_ECW', 'BIA-BIA_ICW', 'BIA-BIA_TBW']].sum(axis=1)\ndf_process['Lean_Body_Mass'] = df_process[['BIA-BIA_FFM', 'BIA-BIA_LDM', 'BIA-BIA_LST']].sum(axis=1)\ndf_process['Muscle_and_Bone_Mass'] = df_process[['BIA-BIA_SMM', 'BIA-BIA_BMC']].sum(axis=1)\n\n# Crear las categorías del tipo de cuerpo y normalizarlas en una sola función\ndf_process['BMI_Category'] = pd.cut(\n    df_process['BIA-BIA_BMI'],\n    bins=[0, 18.5, 24.9, 29.9, float('inf')],\n    labels=[0, 1, 2, 3]  # 0: Underweight, 1: Normal, 2: Overweight, 3: Obese\n)\n\ndf_process = pd.get_dummies(df_process, columns=['BIA-BIA_Frame_num'], prefix='Body_Frame')\n\n# CREAR INDICADORES DE SALUD Y ACTIVIDAD\ndf_process['Metabolic_Efficiency'] = df_process['BIA-BIA_BMR'] / df_process['BIA-BIA_DEE']\ndf_process['Water_to_Lean_Mass_Ratio'] = df_process['BIA-BIA_TBW'] / df_process['Lean_Body_Mass']\n\n# eLIMINAR COLUMNAS QUE SE COMBINARON EN INDICADORES GENERALES\nBio_electric_columns = ['BIA-BIA_ECW', 'BIA-BIA_ICW', 'BIA-BIA_TBW', 'BIA-BIA_FFM', \n                   'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_BMC', \n                   'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_Frame_num']\n\ndf_process = df_process.drop(columns=Bio_electric_columns, axis=1, errors=\"ignore\")\ndf_process","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:22.839775Z","iopub.execute_input":"2024-11-27T17:05:22.840203Z","iopub.status.idle":"2024-11-27T17:05:22.891060Z","shell.execute_reply.started":"2024-11-27T17:05:22.840157Z","shell.execute_reply":"2024-11-27T17:05:22.890019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#ELIMINAR COLUMNAS DE TIEMPO DE PRUEBAS FITNESS Y SOLO DEJAR EL MAXIMO RESULTADO\ndf_process = df_process.drop(columns=['Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec'], axis=1, errors=\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:22.892293Z","iopub.execute_input":"2024-11-27T17:05:22.892642Z","iopub.status.idle":"2024-11-27T17:05:22.898832Z","shell.execute_reply.started":"2024-11-27T17:05:22.892607Z","shell.execute_reply":"2024-11-27T17:05:22.897789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#ELIMINAR COLUMNAS DE FUERZA DE AGARRE\ndf_process = df_process.drop(columns=['Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec'], axis=1, errors=\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:22.906898Z","iopub.execute_input":"2024-11-27T17:05:22.907283Z","iopub.status.idle":"2024-11-27T17:05:22.914287Z","shell.execute_reply.started":"2024-11-27T17:05:22.907248Z","shell.execute_reply":"2024-11-27T17:05:22.913234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Combinar actividad de adolecentes y de niños \n# Definir función para calcular el indicador ponderado por edad\ndef calculate_weighted_activity(row):\n    age = row['Basic_Demos-Age']\n    if age < 13:\n        # Para niños, dar un peso mayor al puntaje infantil\n        weighted_score = (row['PAQ_C-PAQ_C_Total'] * 0.7 if pd.notna(row['PAQ_C-PAQ_C_Total']) else 0) + \\\n                         (row['PAQ_A-PAQ_A_Total'] * 0.3 if pd.notna(row['PAQ_A-PAQ_A_Total']) else 0)\n    else:\n        # Para adolescentes, dar un peso mayor al puntaje adolescente\n        weighted_score = (row['PAQ_C-PAQ_C_Total'] * 0.3 if pd.notna(row['PAQ_C-PAQ_C_Total']) else 0) + \\\n                         (row['PAQ_A-PAQ_A_Total'] * 0.7 if pd.notna(row['PAQ_A-PAQ_A_Total']) else 0)\n    return weighted_score\n\n# Aplicar la función para crear la nueva columna de indicador\ndf_process['Weighted_Activity_Score'] = df_process.apply(calculate_weighted_activity, axis=1)\n\n#ELIMINAR COLUMNAS COMBINADAS EN EL NUEVO INDICADOR\ndf_process = df_process.drop(columns=['PAQ_C-PAQ_C_Total', 'PAQ_A-PAQ_A_Total'], axis=1, errors=\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:22.915742Z","iopub.execute_input":"2024-11-27T17:05:22.916256Z","iopub.status.idle":"2024-11-27T17:05:22.951215Z","shell.execute_reply.started":"2024-11-27T17:05:22.916220Z","shell.execute_reply":"2024-11-27T17:05:22.950220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filtrar el DataFrame para mantener solo las filas donde 'sii' no es nulo\nsupervised_usable = df_process[df_process['sii'].notnull()]\n\n# Calcular el conteo de valores nulos por columna\nmissing_count = (\n    supervised_usable.isnull().sum().reset_index()\n)\n\n# Renombrar las columnas\nmissing_count.columns = ['feature', 'null_count']\n# Ordenar por el conteo de valores nulos\nmissing_count = missing_count.sort_values(by='null_count', ascending=False)\n\n# Split missing_count into two parts\nmissing_count_1 = missing_count.iloc[:len(missing_count) // 2].copy()\nmissing_count_2 = missing_count.iloc[len(missing_count) // 2:].copy()\n\n# Calculate the null ratio for each part\nmissing_count_1.loc[:, 'null_ratio'] = missing_count_1['null_count'] / len(supervised_usable)\nmissing_count_2.loc[:, 'null_ratio'] = missing_count_2['null_count'] / len(supervised_usable)\n\n\n# Crear el gráfico de barras horizontales\nplt.figure(figsize=(6, 10))\nplt.title(f'Valores Faltantes sobre las {len(supervised_usable)} muestras que tienen un objetivo')\nplt.barh(np.arange(len(missing_count_1)), missing_count_1['null_ratio'], color='coral', label='Faltantes')\nplt.barh(np.arange(len(missing_count_1)),\n         1 - missing_count_1['null_ratio'],\n         left=missing_count_1['null_ratio'],\n         color='darkseagreen', label='Disponibles')\nplt.yticks(np.arange(len(missing_count_1)), missing_count_1['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()\n\n# Crear el gráfico de barras horizontales\nplt.figure(figsize=(6, 10))\nplt.title(f'Valores Faltantes sobre las {len(supervised_usable)} muestras que tienen un objetivo')\nplt.barh(np.arange(len(missing_count_2)), missing_count_2['null_ratio'], color='coral', label='Faltantes')\nplt.barh(np.arange(len(missing_count_2)),\n         1 - missing_count_2['null_ratio'],\n         left=missing_count_2['null_ratio'],\n         color='darkseagreen', label='Disponibles')\nplt.yticks(np.arange(len(missing_count_2)), missing_count_2['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:22.952643Z","iopub.execute_input":"2024-11-27T17:05:22.953497Z","iopub.status.idle":"2024-11-27T17:05:24.802558Z","shell.execute_reply.started":"2024-11-27T17:05:22.953447Z","shell.execute_reply":"2024-11-27T17:05:24.801426Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**COMPLETITUD DE DATOS FALTANTES**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\n# Función para imputar valores usando regresión\ndef impute_missing_values(df, target_column):\n    # Filtrar filas sin valores nulos en la columna objetivo\n    df_notnull = df[df[target_column].notnull()]\n    df_null = df[df[target_column].isnull()]\n    \n    # Si hay filas con datos nulos, aplicamos la imputación\n    if not df_null.empty:\n        # Seleccionar las variables predictoras y el objetivo\n        X_train = df_notnull[['Basic_Demos-Age', 'Basic_Demos-Sex', 'Muscle_and_Bone_Mass', 'Weighted_Activity_Score']]\n        y_train = df_notnull[target_column]\n        X_test = df_null[['Basic_Demos-Age', 'Basic_Demos-Sex', 'Muscle_and_Bone_Mass', 'Weighted_Activity_Score']]\n        \n        # Entrenar el modelo de regresión\n        model = LinearRegression()\n        model.fit(X_train, y_train)\n        \n        # Predecir los valores para las filas con datos nulos\n        predicted_values = model.predict(X_test)\n        \n        # Asignar los valores imputados\n        df.loc[df[target_column].isnull(), target_column] = predicted_values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:24.804058Z","iopub.execute_input":"2024-11-27T17:05:24.804424Z","iopub.status.idle":"2024-11-27T17:05:24.811306Z","shell.execute_reply.started":"2024-11-27T17:05:24.804387Z","shell.execute_reply":"2024-11-27T17:05:24.810091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def impute_missing_values_by_age_sex_mean(df, target_column):\n    # Calcular la media por grupo de edad y sexo para la columna objetivo\n    group_means = df.groupby(['Basic_Demos-Age', 'Basic_Demos-Sex'])[target_column].transform('mean')\n\n    # Rellenar los valores nulos en la columna objetivo con la media de su grupo sin usar inplace\n    df[target_column] = df[target_column].fillna(group_means)    ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:24.812617Z","iopub.execute_input":"2024-11-27T17:05:24.812936Z","iopub.status.idle":"2024-11-27T17:05:24.828672Z","shell.execute_reply.started":"2024-11-27T17:05:24.812905Z","shell.execute_reply":"2024-11-27T17:05:24.827705Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Reajustar BIM_Category porque no lo esta reconociendo\n# Convertir 'BMI_Category' a tipo numérico (si es necesario)\ndf_process['BMI_Category'] = pd.to_numeric(df_process['BMI_Category'], errors='coerce')\n\n# Imputar para cada columna con datos faltantes\nimpute_missing_values(df_process, 'Physical-BMI')\nimpute_missing_values(df_process, 'Physical-Height')\nimpute_missing_values(df_process, 'Physical-Weight')\nimpute_missing_values(df_process, 'BIA-BIA_Activity_Level_num')\nimpute_missing_values(df_process, 'Fitness_Endurance-Max_Stage')\nimpute_missing_values(df_process, 'Average_Grip_Strength')\nimpute_missing_values(df_process, 'Average_Sit_Reach')\nimpute_missing_values(df_process, 'Water_to_Lean_Mass_Ratio')\nimpute_missing_values(df_process, 'Metabolic_Efficiency')\nimpute_missing_values(df_process, 'BMI_Category')\n\nimpute_missing_values_by_age_sex_mean(df_process, 'CGAS-CGAS_Score')\n\n\ndf_process\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:24.829846Z","iopub.execute_input":"2024-11-27T17:05:24.830189Z","iopub.status.idle":"2024-11-27T17:05:24.949953Z","shell.execute_reply.started":"2024-11-27T17:05:24.830156Z","shell.execute_reply":"2024-11-27T17:05:24.948904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Seleccionar las columnas con nulos restantes\ncolumns_with_nans = df_process.columns[df_process.isnull().any()].tolist()\n\n# Imputar valores nulos con forward fill sin `inplace=True`\nfor column in columns_with_nans:\n    df_process[column] = df_process[column].fillna(method='ffill')  # Imputación hacia delante","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:24.951168Z","iopub.execute_input":"2024-11-27T17:05:24.951515Z","iopub.status.idle":"2024-11-27T17:05:24.967327Z","shell.execute_reply.started":"2024-11-27T17:05:24.951483Z","shell.execute_reply":"2024-11-27T17:05:24.966316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filtrar el DataFrame para mantener solo las filas donde 'sii' no es nulo\nsupervised_usable = df_process[df_process['sii'].notnull()]\n\n# Calcular el conteo de valores nulos por columna\nmissing_count = (\n    supervised_usable.isnull().sum().reset_index()\n)\n\n# Renombrar las columnas\nmissing_count.columns = ['feature', 'null_count']\n# Ordenar por el conteo de valores nulos\nmissing_count = missing_count.sort_values(by='null_count', ascending=False)\n\n# Split missing_count into two parts\nmissing_count_1 = missing_count.iloc[:len(missing_count) // 2].copy()\nmissing_count_2 = missing_count.iloc[len(missing_count) // 2:].copy()\n\n# Calculate the null ratio for each part\nmissing_count_1.loc[:, 'null_ratio'] = missing_count_1['null_count'] / len(supervised_usable)\nmissing_count_2.loc[:, 'null_ratio'] = missing_count_2['null_count'] / len(supervised_usable)\n\n\n# Crear el gráfico de barras horizontales\nplt.figure(figsize=(6, 10))\nplt.title(f'Valores Faltantes sobre las {len(supervised_usable)} muestras que tienen un objetivo')\nplt.barh(np.arange(len(missing_count_1)), missing_count_1['null_ratio'], color='coral', label='Faltantes')\nplt.barh(np.arange(len(missing_count_1)),\n         1 - missing_count_1['null_ratio'],\n         left=missing_count_1['null_ratio'],\n         color='darkseagreen', label='Disponibles')\nplt.yticks(np.arange(len(missing_count_1)), missing_count_1['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()\n\n# Crear el gráfico de barras horizontales\nplt.figure(figsize=(6, 10))\nplt.title(f'Valores Faltantes sobre las {len(supervised_usable)} muestras que tienen un objetivo')\nplt.barh(np.arange(len(missing_count_2)), missing_count_2['null_ratio'], color='coral', label='Faltantes')\nplt.barh(np.arange(len(missing_count_2)),\n         1 - missing_count_2['null_ratio'],\n         left=missing_count_2['null_ratio'],\n         color='darkseagreen', label='Disponibles')\nplt.yticks(np.arange(len(missing_count_2)), missing_count_2['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:24.968849Z","iopub.execute_input":"2024-11-27T17:05:24.969577Z","iopub.status.idle":"2024-11-27T17:05:26.836614Z","shell.execute_reply.started":"2024-11-27T17:05:24.969530Z","shell.execute_reply":"2024-11-27T17:05:26.835605Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**NORMALIZAR LOS DATOS PARA USAR EN LA SIGUIENTE FASE**","metadata":{}},{"cell_type":"markdown","source":"Procesar variables categoricas","metadata":{}},{"cell_type":"code","source":"# Seleccionar las columnas específicas que deseas vectorizar\ncolumns_to_vectorize = ['Body_Frame_1.0', 'Body_Frame_2.0', 'Body_Frame_3.0']\n\n# Crear la columna vectorizada\nvectorized_column = df_process[columns_to_vectorize].astype(int).apply(lambda x: ''.join(x.astype(str)), axis=1)\n\n# Aplicar One Hot Encoding a la columna vectorizada\none_hot_encoded = pd.get_dummies(vectorized_column, prefix='Body_Frame')\n\n# Crear una nueva columna que contenga listas con los valores de las columnas One Hot Encoded\ndf_process['Body_Frame'] = one_hot_encoded.values.tolist()\n\n# Seleccionar solo la nueva columna y eliminar las originales y las de One Hot Encoding\ndf_process['Body_Frame'] = df_process[['Body_Frame']]\n\n\nprint(df_process)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:26.837884Z","iopub.execute_input":"2024-11-27T17:05:26.838199Z","iopub.status.idle":"2024-11-27T17:05:26.948793Z","shell.execute_reply.started":"2024-11-27T17:05:26.838168Z","shell.execute_reply":"2024-11-27T17:05:26.947797Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_process = df_process.drop(columns=['Body_Frame', 'id'], axis=1, errors=\"ignore\")\ndf_process","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:26.949892Z","iopub.execute_input":"2024-11-27T17:05:26.950189Z","iopub.status.idle":"2024-11-27T17:05:26.982241Z","shell.execute_reply.started":"2024-11-27T17:05:26.950159Z","shell.execute_reply":"2024-11-27T17:05:26.981279Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def normalize(df, target_column):\n    # Convertir los datos de la columna objetivo a un array de numpy\n    data = df[target_column].to_numpy()\n    min_value = data.min()\n    max_value = data.max()\n\n    # Normalizar los valores en la columna objetivo\n    df[target_column] = df[target_column].apply(lambda v: (v - min_value) / (max_value - min_value))\n    \n    # Devolver el DataFrame modificado\n    return df\n\ndef normalize_boleans(df, traget_column):\n    df[traget_column] = df[traget_column].astype(int)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:26.983680Z","iopub.execute_input":"2024-11-27T17:05:26.984402Z","iopub.status.idle":"2024-11-27T17:05:26.990115Z","shell.execute_reply.started":"2024-11-27T17:05:26.984331Z","shell.execute_reply":"2024-11-27T17:05:26.989186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_process)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:26.991556Z","iopub.execute_input":"2024-11-27T17:05:26.991959Z","iopub.status.idle":"2024-11-27T17:05:27.012919Z","shell.execute_reply.started":"2024-11-27T17:05:26.991914Z","shell.execute_reply":"2024-11-27T17:05:27.011969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Remover la columna 'id'\ndf_process = df_process.drop(columns=['id'], errors='ignore')\n\n# Restablecer el índice\ndf_process.reset_index(drop=True, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:27.014197Z","iopub.execute_input":"2024-11-27T17:05:27.014529Z","iopub.status.idle":"2024-11-27T17:05:27.021381Z","shell.execute_reply.started":"2024-11-27T17:05:27.014498Z","shell.execute_reply":"2024-11-27T17:05:27.020408Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\n# Seleccionar solo columnas numéricas sin nulos\nnumeric_columns = df_process.select_dtypes(include=['number']).columns\n#Eliminar la variable objetivo del normalizado \ntarget_column = 'sii'  # Reemplaza 'target' con el nombre de tu variable objetivo\n# Eliminar la columna objetivo de la lista de columnas numéricas\nnumeric_columns = [col for col in numeric_columns if col != target_column]\n\nnumeric_non_null_columns = df_process[numeric_columns].dropna(axis=1).columns\n\n# Crear el escalador Min-Max\nscaler = MinMaxScaler()\n\n# Aplicar el escalador solo a las columnas numéricas sin nulos\ndf_process[numeric_non_null_columns] = scaler.fit_transform(df_process[numeric_non_null_columns])\n\ndf_process\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:27.023141Z","iopub.execute_input":"2024-11-27T17:05:27.023610Z","iopub.status.idle":"2024-11-27T17:05:27.083544Z","shell.execute_reply.started":"2024-11-27T17:05:27.023562Z","shell.execute_reply":"2024-11-27T17:05:27.082597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Seleccionar solo las columnas booleanas\nboolean_columns = df_process.select_dtypes(include=['bool'])\nprint(boolean_columns)\n#Normalizar datos boleanos\nfor column in boolean_columns:\n    df_process = normalize_boleans(df_process, column)\n\ndf_process","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:27.084888Z","iopub.execute_input":"2024-11-27T17:05:27.085283Z","iopub.status.idle":"2024-11-27T17:05:27.120577Z","shell.execute_reply.started":"2024-11-27T17:05:27.085251Z","shell.execute_reply":"2024-11-27T17:05:27.119602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Seleccionar solo las columnas de tipo objeto (categórico)\nstring_columns = df_process.select_dtypes(include=['object'])\nprint(string_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:27.121954Z","iopub.execute_input":"2024-11-27T17:05:27.122679Z","iopub.status.idle":"2024-11-27T17:05:27.130559Z","shell.execute_reply.started":"2024-11-27T17:05:27.122631Z","shell.execute_reply":"2024-11-27T17:05:27.129531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"string_columns = ['FGC-CurlUp_Status', 'FGC-PushUp_Status', 'BMI_Category']\n\nprint(df_process[string_columns].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:27.131707Z","iopub.execute_input":"2024-11-27T17:05:27.131998Z","iopub.status.idle":"2024-11-27T17:05:27.155377Z","shell.execute_reply.started":"2024-11-27T17:05:27.131969Z","shell.execute_reply":"2024-11-27T17:05:27.154255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nstatus_mapping = {\n    \"Needs Improvement\": 0,\n    \"Healthy Fitness Zone\": 1\n}\n\n# Aplicar el mapeo a las columnas específicas\nstring_columns = ['FGC-CurlUp_Status', 'FGC-PushUp_Status']\n\n# Aplicar el mapeo y llenar NaN con un valor predeterminado (por ejemplo, -1 para indicar valores no mapeados)\nfor column in string_columns:\n    df_process[column] = df_process[column].map(status_mapping).fillna(1)  # Puedes usar otro valor si prefieres\n\n\ndf_process[string_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:27.156680Z","iopub.execute_input":"2024-11-27T17:05:27.157027Z","iopub.status.idle":"2024-11-27T17:05:27.174069Z","shell.execute_reply.started":"2024-11-27T17:05:27.156987Z","shell.execute_reply":"2024-11-27T17:05:27.172898Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Eliminar colunmas relacionadas a indices X,Y, Z debido a que enmo las contiene en su calculo, tambien las columnas que indican el día y la temporada debido a que el análisis no se va a enfocar por esos datos y ya se obtuvo la información necesaria para algunos datos compuestos","metadata":{}},{"cell_type":"code","source":"selected_columns = df.filter(regex='^(X_|Y_|Z_|weekday_|quarter_)')\ndf_process = df_process.drop(columns=list(selected_columns.columns), axis=1)\ndf_process","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:27.176106Z","iopub.execute_input":"2024-11-27T17:05:27.176648Z","iopub.status.idle":"2024-11-27T17:05:27.213127Z","shell.execute_reply.started":"2024-11-27T17:05:27.176602Z","shell.execute_reply":"2024-11-27T17:05:27.212084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filtrar el DataFrame para mantener solo las filas donde 'sii' no es nulo\nsupervised_usable = df_process[df_process['sii'].notnull()]\n\n# Calcular el conteo de valores nulos por columna\nmissing_count = (\n    supervised_usable.isnull().sum().reset_index()\n)\n\n# Renombrar las columnas\nmissing_count.columns = ['feature', 'null_count']\n# Ordenar por el conteo de valores nulos\nmissing_count = missing_count.sort_values(by='null_count', ascending=False)\n\n# Split missing_count into two parts\nmissing_count_1 = missing_count.iloc[:len(missing_count) // 2].copy()\nmissing_count_2 = missing_count.iloc[len(missing_count) // 2:].copy()\n\n# Calculate the null ratio for each part\nmissing_count_1.loc[:, 'null_ratio'] = missing_count_1['null_count'] / len(supervised_usable)\nmissing_count_2.loc[:, 'null_ratio'] = missing_count_2['null_count'] / len(supervised_usable)\n\n\n# Crear el gráfico de barras horizontales\nplt.figure(figsize=(6, 10))\nplt.title(f'Valores Faltantes sobre las {len(supervised_usable)} muestras que tienen un objetivo')\nplt.barh(np.arange(len(missing_count_1)), missing_count_1['null_ratio'], color='coral', label='Faltantes')\nplt.barh(np.arange(len(missing_count_1)),\n         1 - missing_count_1['null_ratio'],\n         left=missing_count_1['null_ratio'],\n         color='darkseagreen', label='Disponibles')\nplt.yticks(np.arange(len(missing_count_1)), missing_count_1['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()\n\n# Crear el gráfico de barras horizontales\nplt.figure(figsize=(6, 10))\nplt.title(f'Valores Faltantes sobre las {len(supervised_usable)} muestras que tienen un objetivo')\nplt.barh(np.arange(len(missing_count_2)), missing_count_2['null_ratio'], color='coral', label='Faltantes')\nplt.barh(np.arange(len(missing_count_2)),\n         1 - missing_count_2['null_ratio'],\n         left=missing_count_2['null_ratio'],\n         color='darkseagreen', label='Disponibles')\nplt.yticks(np.arange(len(missing_count_2)), missing_count_2['feature'])\nplt.gca().xaxis.set_major_formatter(PercentFormatter(xmax=1, decimals=0))\nplt.xlim(0, 1)\nplt.legend()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:27.214542Z","iopub.execute_input":"2024-11-27T17:05:27.214969Z","iopub.status.idle":"2024-11-27T17:05:28.479843Z","shell.execute_reply.started":"2024-11-27T17:05:27.214922Z","shell.execute_reply":"2024-11-27T17:05:28.478821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_columns = df_process.select_dtypes(include=['object', 'category']).columns\nprint(\"Columnas categóricas:\", categorical_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:28.481046Z","iopub.execute_input":"2024-11-27T17:05:28.481376Z","iopub.status.idle":"2024-11-27T17:05:28.487679Z","shell.execute_reply.started":"2024-11-27T17:05:28.481327Z","shell.execute_reply":"2024-11-27T17:05:28.486591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Seleccionar solo las columnas numéricas del DataFrame\nnumeric_columns = df_process.select_dtypes(include=['number'])\nnon_numeric_columns = df_process.select_dtypes(exclude=['number'])\nprint(non_numeric_columns.columns)\n\n\n\n# Verificar que todas las columnas numéricas estén en el rango [0, 1]\nis_within_range = numeric_columns.apply(lambda x: x.between(-0.999, 1.0001).all())\n\n# Comprobar si todas las columnas cumplen la condición\nif is_within_range.all():\n    print(\"Todas las columnas numéricas están dentro del rango [0, 1].\")\nelse:\n    print(\"Hay columnas numéricas fuera del rango [0, 1]:\")\n    # Mostrar las columnas numéricas que no están dentro del rango\n    out_of_range_columns = is_within_range[~is_within_range].index\n    print(out_of_range_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:28.488773Z","iopub.execute_input":"2024-11-27T17:05:28.489056Z","iopub.status.idle":"2024-11-27T17:05:28.538612Z","shell.execute_reply.started":"2024-11-27T17:05:28.489028Z","shell.execute_reply":"2024-11-27T17:05:28.537540Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **FASE 4**","metadata":{}},{"cell_type":"markdown","source":"Esta fase intentará abordar de manera general el modelado de los datos con base a la vista minable, para ello si implementaran los siguientes modelos:\n\n\n* AdaBoost\n* GradientBoost\n* XBoost\n* Kmean\n* Kmean++\n* Kmean Mnahatan\n* RandomForest\n\n\nEstos modelos se realizaran de forma general sin profundizar en los metadatos y en la posibles variaciones que puedan posiblemente llegar a mejorar los resultados, esstos porque del plano general se selecciona el mejor de los modelos generales, a partir de este modelo se profundizará en la busqueda de la mejora del modelado.","metadata":{}},{"cell_type":"code","source":"# Separar las características y la variable objetivo\ndf_model = df_process.drop(columns=['sii'])\ntarget = df_process['sii']\n\n# Dividir el conjunto de datos en entrenamiento y prueba\nX_train, X_test, y_train, y_test = train_test_split(df_model, target, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:28.539916Z","iopub.execute_input":"2024-11-27T17:05:28.540246Z","iopub.status.idle":"2024-11-27T17:05:28.557267Z","shell.execute_reply.started":"2024-11-27T17:05:28.540215Z","shell.execute_reply":"2024-11-27T17:05:28.556282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# **Revisar distribución de las clases en el conjunto de prueba** (sin aplicar SMOTE)\nprint(f'Distribución de y_test (sin balanceo): {Counter(y_train)}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:28.558466Z","iopub.execute_input":"2024-11-27T17:05:28.558794Z","iopub.status.idle":"2024-11-27T17:05:28.564927Z","shell.execute_reply.started":"2024-11-27T17:05:28.558756Z","shell.execute_reply":"2024-11-27T17:05:28.563876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_model = df_process.drop(columns=['sii'])\ntarget = df_process['sii']\n\nX_train, X_test, y_train, y_test = train_test_split(df_model, target, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:28.566262Z","iopub.execute_input":"2024-11-27T17:05:28.566727Z","iopub.status.idle":"2024-11-27T17:05:28.584202Z","shell.execute_reply.started":"2024-11-27T17:05:28.566680Z","shell.execute_reply":"2024-11-27T17:05:28.583149Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Antes de usar los algoritmos de boosting se usa validación cruzada para encontrar el mejor valor de estimadores (eslabones débiles) para realizar el proceso de modelado","metadata":{}},{"cell_type":"markdown","source":"# **GRADIENT BOOSTING MACHINE - GBM**","metadata":{}},{"cell_type":"code","source":"from sklearn import ensemble\n\n# Fit classifier with out-of-bag estimates\nparams = {\n    \"n_estimators\": 1200,\n    \"max_depth\": 3,\n    \"subsample\": 0.5,\n    \"learning_rate\": 0.01,\n    \"min_samples_leaf\": 1,\n    \"random_state\": 3,\n}\ngbm = ensemble.GradientBoostingClassifier(**params)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:28.585931Z","iopub.execute_input":"2024-11-27T17:05:28.586371Z","iopub.status.idle":"2024-11-27T17:05:28.592196Z","shell.execute_reply.started":"2024-11-27T17:05:28.586306Z","shell.execute_reply":"2024-11-27T17:05:28.591137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:28.593441Z","iopub.execute_input":"2024-11-27T17:05:28.593751Z","iopub.status.idle":"2024-11-27T17:05:28.611664Z","shell.execute_reply.started":"2024-11-27T17:05:28.593720Z","shell.execute_reply":"2024-11-27T17:05:28.610575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:28.612985Z","iopub.execute_input":"2024-11-27T17:05:28.613467Z","iopub.status.idle":"2024-11-27T17:05:28.627326Z","shell.execute_reply.started":"2024-11-27T17:05:28.613433Z","shell.execute_reply":"2024-11-27T17:05:28.626131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Entrenar el modelo\ngbm.fit(X_train, y_train)\n\n# Predecir las etiquetas para el conjunto de prueba\ny_pred = gbm.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:28.628688Z","iopub.execute_input":"2024-11-27T17:05:28.629038Z","iopub.status.idle":"2024-11-27T17:05:59.759765Z","shell.execute_reply.started":"2024-11-27T17:05:28.629005Z","shell.execute_reply":"2024-11-27T17:05:59.758671Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular el accuracy\naccuracy = gbm.score(X_test, y_test)\nprint(f\"Accuracy: {accuracy:.4f}\")\n\n# Obtener el reporte de clasificación\nreport = classification_report(y_test, y_pred, output_dict=True)\n\n# Convertir el reporte de clasificación a un DataFrame para mostrarlo en forma de tabla\nreport_df = pd.DataFrame(report).transpose()\n\n# Mostrar el DataFrame con las métricas\nprint(\"\\nReporte de clasificación GRADIENT BOOSTING MACHINE:\\n\")\nprint(report_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:59.760996Z","iopub.execute_input":"2024-11-27T17:05:59.761326Z","iopub.status.idle":"2024-11-27T17:05:59.794778Z","shell.execute_reply.started":"2024-11-27T17:05:59.761294Z","shell.execute_reply":"2024-11-27T17:05:59.793763Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:05:59.796340Z","iopub.execute_input":"2024-11-27T17:05:59.797275Z","iopub.status.idle":"2024-11-27T17:06:00.068364Z","shell.execute_reply.started":"2024-11-27T17:05:59.797226Z","shell.execute_reply":"2024-11-27T17:06:00.067301Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"El modelo de GBM nos indica una clasificación perfecta de todos los casos en el conjunto de pruebas, esto puede parecer perfecto, pero hay que tener en cuenta que el dataset esta muy desbalanceado, con lo cual para las categorías 2 y 3 este reporte puede cambiar si se llegaran a balancear, hay muy pocos datos en generar de nuestro dataset, con solo 996 datos registrados, a esto agregar el limitante de las las clases 2 y 3 que si se comparan la diferencia de datos es controversial.","metadata":{}},{"cell_type":"markdown","source":"# XGBoost","metadata":{}},{"cell_type":"code","source":"dtrain = xgb.DMatrix(X_train, label=y_train)\ndtest = xgb.DMatrix(X_test, label=y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:00.069554Z","iopub.execute_input":"2024-11-27T17:06:00.069880Z","iopub.status.idle":"2024-11-27T17:06:00.113862Z","shell.execute_reply.started":"2024-11-27T17:06:00.069849Z","shell.execute_reply":"2024-11-27T17:06:00.113047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n    'objective': 'multi:softmax',\n    'num_class': len(np.unique(y_train)),\n    'eval_metric': 'merror'\n}\n\n# Entrenar modelo XGBoost\nxgb_model = xgb.train(params, dtrain, num_boost_round=100)\n\n# Realizar predicciones con XGBoost\ny_pred = xgb_model.predict(dtest)\n\n# Convertir las predicciones a enteros (ya que son clases)\ny_pred = np.array([int(x) for x in y_pred])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:00.114755Z","iopub.execute_input":"2024-11-27T17:06:00.115065Z","iopub.status.idle":"2024-11-27T17:06:00.408016Z","shell.execute_reply.started":"2024-11-27T17:06:00.115032Z","shell.execute_reply":"2024-11-27T17:06:00.407221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular el accuracy para XGBoost\naccuracy_xgb = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy XGBoost: {accuracy_xgb:.4f}\")\n\n# Obtener el reporte de clasificación para XGBoost\nreport_xgb = classification_report(y_test, y_pred, output_dict=True)\n\n# Convertir el reporte de clasificación a un DataFrame para mostrarlo en forma de tabla\nreport_xgb_df = pd.DataFrame(report_xgb).transpose()\n\n# Mostrar el reporte para XGBoost\nprint(\"\\nReporte de clasificación XGBoost:\\n\")\nprint(report_xgb_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:00.408912Z","iopub.execute_input":"2024-11-27T17:06:00.409204Z","iopub.status.idle":"2024-11-27T17:06:00.427930Z","shell.execute_reply.started":"2024-11-27T17:06:00.409173Z","shell.execute_reply":"2024-11-27T17:06:00.427266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues', cbar=False, \n            xticklabels=np.unique(y_test), yticklabels=np.unique(y_test))\n\nplt.title('Matriz de Confusión NGBoost')\nplt.xlabel('Predicciones')\nplt.ylabel('Valores Reales')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:00.428750Z","iopub.execute_input":"2024-11-27T17:06:00.429036Z","iopub.status.idle":"2024-11-27T17:06:00.627444Z","shell.execute_reply.started":"2024-11-27T17:06:00.429007Z","shell.execute_reply":"2024-11-27T17:06:00.626422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **RANDOM FOREST**","metadata":{}},{"cell_type":"code","source":"model = RandomForestClassifier(class_weight='balanced')\nmodel.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:00.628720Z","iopub.execute_input":"2024-11-27T17:06:00.629151Z","iopub.status.idle":"2024-11-27T17:06:01.126874Z","shell.execute_reply.started":"2024-11-27T17:06:00.629106Z","shell.execute_reply":"2024-11-27T17:06:01.125877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:01.128063Z","iopub.execute_input":"2024-11-27T17:06:01.128421Z","iopub.status.idle":"2024-11-27T17:06:01.149324Z","shell.execute_reply.started":"2024-11-27T17:06:01.128382Z","shell.execute_reply":"2024-11-27T17:06:01.148272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generar y mostrar el reporte de clasificación\nprint(\"Reporte de clasificación:\\n\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:01.150619Z","iopub.execute_input":"2024-11-27T17:06:01.150949Z","iopub.status.idle":"2024-11-27T17:06:01.168654Z","shell.execute_reply.started":"2024-11-27T17:06:01.150917Z","shell.execute_reply":"2024-11-27T17:06:01.167564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:01.170045Z","iopub.execute_input":"2024-11-27T17:06:01.170482Z","iopub.status.idle":"2024-11-27T17:06:01.413985Z","shell.execute_reply.started":"2024-11-27T17:06:01.170436Z","shell.execute_reply":"2024-11-27T17:06:01.412932Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"En el modelo de random forest se ve reflejado el problema con el balance de los datos, a diferendia de gbm, en RF no se logran clasificar satisfactoriamente las clases 2 y 3, esto es debido a que la cantidad de datos es muy poca y afecta en la clasificación, esto puede ser un modelado más real respecto al comportamiento de los datos, aunque las métricas sean peores estas nos ayudan a tener en cuenta la densidad de datos de las clases 2 y 3","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# **KNN**","metadata":{}},{"cell_type":"markdown","source":"**Antes de usar como tal el modelo knn, este recibe un parámetro importante que es el número de vecinos, para seleccionar un número óptimo se usará validación cruzada para encontrar el mejor k.**","metadata":{}},{"cell_type":"code","source":"knn = KNeighborsClassifier()\n\n# rango de valores de K\nparam_grid = {'n_neighbors': range(1, 21)}\n\n# GridSearchCV con validación cruzada de 5 pliegues\ngrid_search = GridSearchCV(estimator=knn, param_grid=param_grid, cv=5, scoring='accuracy')\n\n# Entrenar con los datos\ngrid_search.fit(X_train, y_train)\n\n# Obtener el mejor valor de K y el mejor rendimiento\nbest_k = grid_search.best_params_['n_neighbors']\nbest_score = grid_search.best_score_\n\nprint(f\"Mejor valor de K: {best_k}\")\nprint(f\"Mejor accuracy en validación cruzada: {best_score:.4f}\")\n\n# Probar el modelo con el mejor valor de K en el conjunto de prueba\nbest_knn = grid_search.best_estimator_\ny_pred = best_knn.predict(X_test)\ntest_accuracy = accuracy_score(y_test, y_pred)\n\nprint(f\"Accuracy en el conjunto de prueba con el mejor K ({best_k}): {test_accuracy:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:01.415442Z","iopub.execute_input":"2024-11-27T17:06:01.415891Z","iopub.status.idle":"2024-11-27T17:06:03.976032Z","shell.execute_reply.started":"2024-11-27T17:06:01.415846Z","shell.execute_reply":"2024-11-27T17:06:03.974900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors=11)\nknn.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:03.977283Z","iopub.execute_input":"2024-11-27T17:06:03.977649Z","iopub.status.idle":"2024-11-27T17:06:03.992416Z","shell.execute_reply.started":"2024-11-27T17:06:03.977616Z","shell.execute_reply":"2024-11-27T17:06:03.991269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = knn.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:03.993885Z","iopub.execute_input":"2024-11-27T17:06:03.994720Z","iopub.status.idle":"2024-11-27T17:06:04.017468Z","shell.execute_reply.started":"2024-11-27T17:06:03.994681Z","shell.execute_reply":"2024-11-27T17:06:04.016500Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generar y mostrar el reporte de clasificación\nprint(\"Reporte de clasificación:\\n\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:04.018892Z","iopub.execute_input":"2024-11-27T17:06:04.019396Z","iopub.status.idle":"2024-11-27T17:06:04.034874Z","shell.execute_reply.started":"2024-11-27T17:06:04.019319Z","shell.execute_reply":"2024-11-27T17:06:04.033792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:04.036145Z","iopub.execute_input":"2024-11-27T17:06:04.036472Z","iopub.status.idle":"2024-11-27T17:06:04.304753Z","shell.execute_reply.started":"2024-11-27T17:06:04.036440Z","shell.execute_reply":"2024-11-27T17:06:04.303730Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Para el caso de KNN la clasificación empieza a afector a todas las clases, apesar de que se hace un tunning muy simple en el número de vecinos esto no tiene un impacto significativo en la clasificación que podría ser que se ve afectada por la densidad de datos.","metadata":{}},{"cell_type":"markdown","source":"# **SUPPORT VECTOR MACHINE - SVM**","metadata":{}},{"cell_type":"code","source":"clf = SVC(kernel='linear', class_weight='balanced')\n\n# Entrenar el modelo\nclf.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:04.305862Z","iopub.execute_input":"2024-11-27T17:06:04.306161Z","iopub.status.idle":"2024-11-27T17:06:04.346273Z","shell.execute_reply.started":"2024-11-27T17:06:04.306131Z","shell.execute_reply":"2024-11-27T17:06:04.345207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Realizar predicciones\ny_pred = clf.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:04.347526Z","iopub.execute_input":"2024-11-27T17:06:04.347832Z","iopub.status.idle":"2024-11-27T17:06:04.362286Z","shell.execute_reply.started":"2024-11-27T17:06:04.347801Z","shell.execute_reply":"2024-11-27T17:06:04.361237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generar y mostrar el reporte de clasificación\nprint(\"Reporte de clasificación:\\n\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:04.363677Z","iopub.execute_input":"2024-11-27T17:06:04.364075Z","iopub.status.idle":"2024-11-27T17:06:04.379156Z","shell.execute_reply.started":"2024-11-27T17:06:04.364041Z","shell.execute_reply":"2024-11-27T17:06:04.378023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:04.380637Z","iopub.execute_input":"2024-11-27T17:06:04.381051Z","iopub.status.idle":"2024-11-27T17:06:04.602376Z","shell.execute_reply.started":"2024-11-27T17:06:04.381005Z","shell.execute_reply":"2024-11-27T17:06:04.601324Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### AHORA SE PROBARAN LOS 5 MODELOS NATERIORES PERO CON UN AJUSTE DE BALANCEO COMO TUNNING POR BAGGIN Y AJUSTE DE HIPERPARAMETROS QUE ALGUNOS MODELOS NOS OFRECEM","metadata":{}},{"cell_type":"markdown","source":"##### Para realizar el balanceo de datos hay que tener en cuenta que serán datos sintéticos, aunque en la p´rueba de kaggle se ofrecen muchos datos, solo ofrecen unos limitados para el conjunto de archivos actigráficos, por ello se opta por hacer un balanceo de forma sintética","metadata":{}},{"cell_type":"code","source":"# Aplicar SMOTE para el balanceo\nsmote = SMOTE(random_state=42)\nX_train_balanced, y_train_balanced = smote.fit_resample(X_train, y_train)\n\nprint(f\"Antes del balanceo: {dict(zip(*np.unique(y_train, return_counts=True)))}\")\nprint(f\"Después del balanceo: {dict(zip(*np.unique(y_train_balanced, return_counts=True)))}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:04.603560Z","iopub.execute_input":"2024-11-27T17:06:04.603858Z","iopub.status.idle":"2024-11-27T17:06:04.634840Z","shell.execute_reply.started":"2024-11-27T17:06:04.603829Z","shell.execute_reply":"2024-11-27T17:06:04.633898Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### GRADIENT BOOSTING MACHINE - GBM","metadata":{}},{"cell_type":"code","source":"# Fit classifier with out-of-bag estimates\nparams = {\n    \"n_estimators\": 1200,\n    \"max_depth\": 3,\n    \"subsample\": 0.5,\n    \"learning_rate\": 0.01,\n    \"min_samples_leaf\": 1,\n    \"random_state\": 3,\n}\ngbm = ensemble.GradientBoostingClassifier(**params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:04.636174Z","iopub.execute_input":"2024-11-27T17:06:04.636825Z","iopub.status.idle":"2024-11-27T17:06:04.643830Z","shell.execute_reply.started":"2024-11-27T17:06:04.636777Z","shell.execute_reply":"2024-11-27T17:06:04.642816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Entrenar el modelo\ngbm.fit(X_train_balanced, y_train_balanced)\n\n# Predecir las etiquetas para el conjunto de prueba\ny_pred = gbm.predict(X_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:06:04.645806Z","iopub.execute_input":"2024-11-27T17:06:04.646144Z","iopub.status.idle":"2024-11-27T17:07:17.532960Z","shell.execute_reply.started":"2024-11-27T17:06:04.646114Z","shell.execute_reply":"2024-11-27T17:07:17.532061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular el accuracy\naccuracy = gbm.score(X_test, y_test)\nprint(f\"Accuracy: {accuracy:.4f}\")\n\n# Obtener el reporte de clasificación\nreport = classification_report(y_test, y_pred, output_dict=True)\n\n# Convertir el reporte de clasificación a un DataFrame para mostrarlo en forma de tabla\nreport_df = pd.DataFrame(report).transpose()\n\n# Mostrar el DataFrame con las métricas\nprint(\"\\nReporte de clasificación GRADIENT BOOSTING MACHINE:\\n\")\nprint(report_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:17.534048Z","iopub.execute_input":"2024-11-27T17:07:17.534344Z","iopub.status.idle":"2024-11-27T17:07:17.566788Z","shell.execute_reply.started":"2024-11-27T17:07:17.534315Z","shell.execute_reply":"2024-11-27T17:07:17.565762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:17.568182Z","iopub.execute_input":"2024-11-27T17:07:17.568543Z","iopub.status.idle":"2024-11-27T17:07:17.784489Z","shell.execute_reply.started":"2024-11-27T17:07:17.568510Z","shell.execute_reply":"2024-11-27T17:07:17.783409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### XGBOOST","metadata":{}},{"cell_type":"code","source":"dtrain = xgb.DMatrix(X_train_balanced, label=y_train_balanced)\ndtest = xgb.DMatrix(X_test, label=y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:17.785837Z","iopub.execute_input":"2024-11-27T17:07:17.786249Z","iopub.status.idle":"2024-11-27T17:07:17.820646Z","shell.execute_reply.started":"2024-11-27T17:07:17.786205Z","shell.execute_reply":"2024-11-27T17:07:17.819792Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"params = {\n    'objective': 'multi:softmax',\n    'num_class': len(np.unique(y_train)),\n    'eval_metric': 'merror'\n}\n\n# Entrenar modelo XGBoost\nxgb_model = xgb.train(params, dtrain, num_boost_round=100)\n\n# Realizar predicciones con XGBoost\ny_pred = xgb_model.predict(dtest)\n\n# Convertir las predicciones a enteros (ya que son clases)\ny_pred = np.array([int(x) for x in y_pred])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:17.831726Z","iopub.execute_input":"2024-11-27T17:07:17.832368Z","iopub.status.idle":"2024-11-27T17:07:18.213239Z","shell.execute_reply.started":"2024-11-27T17:07:17.832316Z","shell.execute_reply":"2024-11-27T17:07:18.212411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular el accuracy para XGBoost\naccuracy_xgb = accuracy_score(y_test, y_pred)\nprint(f\"Accuracy XGBoost: {accuracy_xgb:.4f}\")\n\n# Obtener el reporte de clasificación para XGBoost\nreport_xgb = classification_report(y_test, y_pred, output_dict=True)\n\n# Convertir el reporte de clasificación a un DataFrame para mostrarlo en forma de tabla\nreport_xgb_df = pd.DataFrame(report_xgb).transpose()\n\n# Mostrar el reporte para XGBoost\nprint(\"\\nReporte de clasificación XGBoost:\\n\")\nprint(report_xgb_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:18.214160Z","iopub.execute_input":"2024-11-27T17:07:18.214479Z","iopub.status.idle":"2024-11-27T17:07:18.235798Z","shell.execute_reply.started":"2024-11-27T17:07:18.214445Z","shell.execute_reply":"2024-11-27T17:07:18.234722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues', cbar=False, \n            xticklabels=np.unique(y_test), yticklabels=np.unique(y_test))\n\nplt.title('Matriz de Confusión NGBoost')\nplt.xlabel('Predicciones')\nplt.ylabel('Valores Reales')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:18.237100Z","iopub.execute_input":"2024-11-27T17:07:18.237543Z","iopub.status.idle":"2024-11-27T17:07:18.438656Z","shell.execute_reply.started":"2024-11-27T17:07:18.237487Z","shell.execute_reply":"2024-11-27T17:07:18.437432Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Random forest","metadata":{}},{"cell_type":"code","source":"model = RandomForestClassifier(class_weight='balanced')\nmodel.fit(X_train_balanced, y_train_balanced)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:18.440221Z","iopub.execute_input":"2024-11-27T17:07:18.441036Z","iopub.status.idle":"2024-11-27T17:07:19.422498Z","shell.execute_reply.started":"2024-11-27T17:07:18.440983Z","shell.execute_reply":"2024-11-27T17:07:19.421449Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:19.423778Z","iopub.execute_input":"2024-11-27T17:07:19.424107Z","iopub.status.idle":"2024-11-27T17:07:19.443479Z","shell.execute_reply.started":"2024-11-27T17:07:19.424075Z","shell.execute_reply":"2024-11-27T17:07:19.442426Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generar y mostrar el reporte de clasificación\nprint(\"Reporte de clasificación:\\n\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:19.444670Z","iopub.execute_input":"2024-11-27T17:07:19.444986Z","iopub.status.idle":"2024-11-27T17:07:19.459683Z","shell.execute_reply.started":"2024-11-27T17:07:19.444955Z","shell.execute_reply":"2024-11-27T17:07:19.458709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:19.460990Z","iopub.execute_input":"2024-11-27T17:07:19.461440Z","iopub.status.idle":"2024-11-27T17:07:19.724253Z","shell.execute_reply.started":"2024-11-27T17:07:19.461382Z","shell.execute_reply":"2024-11-27T17:07:19.723238Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### KNN","metadata":{}},{"cell_type":"code","source":"# rango de valores de K\nparam_grid = {'n_neighbors': range(1, 21)}\n\n# GridSearchCV con validación cruzada de 5 pliegues\ngrid_search = GridSearchCV(estimator=knn, param_grid=param_grid, cv=5, scoring='accuracy')\n\n# Entrenar con los datos\ngrid_search.fit(X_train, y_train)\n\n# Obtener el mejor valor de K y el mejor rendimiento\nbest_k = grid_search.best_params_['n_neighbors']\nbest_score = grid_search.best_score_\n\nprint(f\"Mejor valor de K: {best_k}\")\nprint(f\"Mejor accuracy en validación cruzada: {best_score:.4f}\")\n\n# Probar el modelo con el mejor valor de K en el conjunto de prueba\nbest_knn = grid_search.best_estimator_\ny_pred = best_knn.predict(X_test)\ntest_accuracy = accuracy_score(y_test, y_pred)\n\nprint(f\"Accuracy en el conjunto de prueba con el mejor K ({best_k}): {test_accuracy:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:19.725673Z","iopub.execute_input":"2024-11-27T17:07:19.726101Z","iopub.status.idle":"2024-11-27T17:07:22.100295Z","shell.execute_reply.started":"2024-11-27T17:07:19.726053Z","shell.execute_reply":"2024-11-27T17:07:22.099209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"knn = KNeighborsClassifier(n_neighbors=11)\nknn.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.101412Z","iopub.execute_input":"2024-11-27T17:07:22.101717Z","iopub.status.idle":"2024-11-27T17:07:22.114645Z","shell.execute_reply.started":"2024-11-27T17:07:22.101687Z","shell.execute_reply":"2024-11-27T17:07:22.113624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = knn.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.116296Z","iopub.execute_input":"2024-11-27T17:07:22.116755Z","iopub.status.idle":"2024-11-27T17:07:22.144145Z","shell.execute_reply.started":"2024-11-27T17:07:22.116706Z","shell.execute_reply":"2024-11-27T17:07:22.143095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generar y mostrar el reporte de clasificación\nprint(\"Reporte de clasificación:\\n\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.145310Z","iopub.execute_input":"2024-11-27T17:07:22.145654Z","iopub.status.idle":"2024-11-27T17:07:22.161035Z","shell.execute_reply.started":"2024-11-27T17:07:22.145622Z","shell.execute_reply":"2024-11-27T17:07:22.160025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.162242Z","iopub.execute_input":"2024-11-27T17:07:22.162574Z","iopub.status.idle":"2024-11-27T17:07:22.428618Z","shell.execute_reply.started":"2024-11-27T17:07:22.162542Z","shell.execute_reply":"2024-11-27T17:07:22.427643Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### SVM","metadata":{}},{"cell_type":"code","source":"clf = SVC(kernel='linear', class_weight='balanced')\n\n# Entrenar el modelo\nclf.fit(X_train_balanced, y_train_balanced)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.429784Z","iopub.execute_input":"2024-11-27T17:07:22.430089Z","iopub.status.idle":"2024-11-27T17:07:22.522931Z","shell.execute_reply.started":"2024-11-27T17:07:22.430059Z","shell.execute_reply":"2024-11-27T17:07:22.521963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Realizar predicciones\ny_pred = clf.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.524148Z","iopub.execute_input":"2024-11-27T17:07:22.524485Z","iopub.status.idle":"2024-11-27T17:07:22.541188Z","shell.execute_reply.started":"2024-11-27T17:07:22.524453Z","shell.execute_reply":"2024-11-27T17:07:22.540035Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generar y mostrar el reporte de clasificación\nprint(\"Reporte de clasificación:\\n\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.542456Z","iopub.execute_input":"2024-11-27T17:07:22.542817Z","iopub.status.idle":"2024-11-27T17:07:22.557193Z","shell.execute_reply.started":"2024-11-27T17:07:22.542783Z","shell.execute_reply":"2024-11-27T17:07:22.556002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.558461Z","iopub.execute_input":"2024-11-27T17:07:22.558884Z","iopub.status.idle":"2024-11-27T17:07:22.826575Z","shell.execute_reply.started":"2024-11-27T17:07:22.558840Z","shell.execute_reply":"2024-11-27T17:07:22.825454Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### CONCLUSIÓN\nEn este caso el balanceo empeora los resultados de algunos de los modelos pero por el contrario gboost y xgb mantienen sus metricas perfectas, esto puede cambiar en el momento de hacer la evaluación contra las pruebas del reto de kaggle por lo cual hasta que no se comparen y se haga un tipo de muestreo contra datos de kaggle reales para pruebas no se sabrá de manera real si el modelo básico seleccionado en fase 4 sirve para cumplir con el objetivo del proyecto.\n\nComo se menciona gb y xgb tienen metricas iguales donde se realizan clasificaciones perfectas por ello la selección entre estos modelos se hace en base a el contexto y las ventajas que nos brinda cada modelo, en este caso xgboost se selecciona porque maneja mejor el uso de recursos computacionales, además el dataset final de training puede que tenga muy pocos datos pero la cantidad de columnas es un factor a considerar, xgboost es más eficiente para datasets grandes por lo cual también es un factor favorable en este caso.\n\nLos modelos implementados estan en su forma más básica, es decir no se hace un ajuste de hiperparametros avanzado o detallado que puede hacer que el modelo mejore mucho, esto porque en fase 4 se evalua cual es el mejor candidato y en fase 5 este modelo se refinará para explotarlo de la mejor manera.","metadata":{}},{"cell_type":"markdown","source":"# FASE 5 - EVALUACIÓN \n\n### Para la fase 5 se procede con la evaluacion del modelo de XGBOOST pero esta vez se hace un refinamiento en el modelado apl icando técnicas más avanzadas para mejorar lo más posible el modelo y luego evaluar los resultados","metadata":{}},{"cell_type":"markdown","source":"### Definicion final del modelado","metadata":{}},{"cell_type":"code","source":"dtrain = xgb.DMatrix(X_train, label=y_train)\ndtest = xgb.DMatrix(X_test, label=y_test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.827845Z","iopub.execute_input":"2024-11-27T17:07:22.828174Z","iopub.status.idle":"2024-11-27T17:07:22.908558Z","shell.execute_reply.started":"2024-11-27T17:07:22.828141Z","shell.execute_reply":"2024-11-27T17:07:22.907438Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### ESTABLECEMOS PARAMETROS INICIALES\n\n* eta (learning_rate): Baja el valor por defecto (0.3) a algo más conservador, como 0.1 o 0.05, para permitir un entrenamiento más gradual.\n* max_depth: Controla la profundidad máxima de los árboles. Un valor entre 4 y 8 suele funcionar bien.\n* subsample: Fracción de datos de entrenamiento utilizados en cada iteración. Usualmente, valores entre 0.7 y 0.9 previenen sobreajuste.\n* colsample_bytree: Fracción de características consideradas en cada árbol. Ajusta entre 0.7 y 0.9.\n* min_child_weight: Controla la cantidad mínima de peso necesario para dividir un nodo. Aumentar este valor puede ayudar a evitar divisiones con poca información.\n* gamma: Penalización por hacer divisiones adicionales\n","metadata":{}},{"cell_type":"code","source":"initial_params = {\n    'objective': 'multi:softmax',\n    'num_class': len(np.unique(y_train)),\n    'eval_metric': 'merror',\n    'eta': 0.1,  # Tasa de aprendizaje más alta\n    'max_depth': 3,  # Árboles más pequeños\n    'subsample': 0.8,  # Usar solo el 80% de los datos por cada árbol\n    'colsample_bytree': 0.8,  # Usar solo el 80% de las características\n    'min_child_weight': 1,  # Menos regularización\n    'gamma': 0.1,\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.909757Z","iopub.execute_input":"2024-11-27T17:07:22.910143Z","iopub.status.idle":"2024-11-27T17:07:22.918382Z","shell.execute_reply.started":"2024-11-27T17:07:22.910103Z","shell.execute_reply":"2024-11-27T17:07:22.917602Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"###  GENERAMOS EL MODELO INICIAL","metadata":{}},{"cell_type":"code","source":"# Crear un modelo básico de XGBoost para búsqueda\nxgb_clf = XGBClassifier(use_label_encoder=False, eval_metric='merror')\n\n# Definir la cuadrícula de búsqueda para mejora de paramatros iniciales\nparam_grid = {\n    'max_depth': [4, 6],\n    'learning_rate': [0.05, .0],\n    'subsample': [0.7, 0.8],\n    'colsample_bytree': [0.7, 0.8],\n    'n_estimators': [50, 100],\n    'gamma': [0, 0.1],\n}\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.919187Z","iopub.execute_input":"2024-11-27T17:07:22.919523Z","iopub.status.idle":"2024-11-27T17:07:22.934546Z","shell.execute_reply.started":"2024-11-27T17:07:22.919490Z","shell.execute_reply":"2024-11-27T17:07:22.933154Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### USAMOS GRID SEARCH PARA VER LOS MEJORES HIPERPARAMETROS","metadata":{}},{"cell_type":"code","source":"# Configurar la búsqueda\ngrid_search = GridSearchCV(estimator=xgb_clf, param_grid=param_grid, scoring='accuracy', cv=3)\ngrid_search.fit(X_train, y_train)\n\n# Obtener los mejores hiperparámetros\nbest_params = grid_search.best_params_\nprint(\"Mejores hiperparámetros encontrados:\", best_params)\n\n# Actualizar los parámetros con los mejores encontrados\nparams = {**initial_params, **best_params}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:07:22.935927Z","iopub.execute_input":"2024-11-27T17:07:22.936337Z","iopub.status.idle":"2024-11-27T17:09:14.743806Z","shell.execute_reply.started":"2024-11-27T17:07:22.936287Z","shell.execute_reply":"2024-11-27T17:09:14.742590Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Para evitar el sobreajuste o intentar controlarlo y tener mejores resultado se usa la técnica de early stopper para controlar esto","metadata":{}},{"cell_type":"code","source":"# Configurar los conjuntos de evaluación\nevals = [(dtrain, 'train'), (dtest, 'eval')]\n\n# Entrenar el modelo con early stopping\nxgb_model = xgb.train(\n    params,\n    dtrain,\n    num_boost_round=1000,\n    evals=evals,\n    early_stopping_rounds=50,\n    verbose_eval=10  # Mostrar métricas cada 10 iteraciones\n)\n\n# Obtener el mejor número de iteraciones\nprint(f\"Mejor número de iteraciones: {xgb_model.best_iteration}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:09:14.745037Z","iopub.execute_input":"2024-11-27T17:09:14.745387Z","iopub.status.idle":"2024-11-27T17:09:15.205597Z","shell.execute_reply.started":"2024-11-27T17:09:14.745337Z","shell.execute_reply":"2024-11-27T17:09:15.204587Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Luego de este refinamiento de hiperparámetros evaluamos la importancia de cada uno para ver en realidad el aporte de cada ajuste, las caracterpisticas que no presenten relevancia en el modelo se eliminan ","metadata":{}},{"cell_type":"code","source":"# Visualizar importancia de características\nxgb.plot_importance(xgb_model, importance_type='weight', max_num_features=10)  # Top 10\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:42:44.505209Z","iopub.execute_input":"2024-11-27T17:42:44.506071Z","iopub.status.idle":"2024-11-27T17:42:44.729666Z","shell.execute_reply.started":"2024-11-27T17:42:44.506036Z","shell.execute_reply":"2024-11-27T17:42:44.728645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Obtener las importancias de características directamente del modelo\nimportance = xgb_model.get_score(importance_type='weight')\n\n# Ordenar las características por importancia y seleccionar las top 10\ntop_features = sorted(importance.items(), key=lambda x: x[1], reverse=True)[:10]\n\n# Extraer solo los nombres de las características\ntop_feature_names = [feature[0] for feature in top_features]\n\n# Mostrar la lista de nombres\nprint(\"Top 10 características importantes:\", top_feature_names)\n\n# Visualizar la importancia (opcional)\nxgb.plot_importance(xgb_model, importance_type='weight', max_num_features=10)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:49:05.694451Z","iopub.execute_input":"2024-11-27T17:49:05.694859Z","iopub.status.idle":"2024-11-27T17:49:05.921285Z","shell.execute_reply.started":"2024-11-27T17:49:05.694812Z","shell.execute_reply":"2024-11-27T17:49:05.920411Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Evaluación simple del modelo","metadata":{}},{"cell_type":"code","source":"# Predicciones en el conjunto de prueba\ny_pred = xgb_model.predict(dtest)\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Precisión del modelo: {accuracy:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:09:15.477529Z","iopub.execute_input":"2024-11-27T17:09:15.477979Z","iopub.status.idle":"2024-11-27T17:09:15.489368Z","shell.execute_reply.started":"2024-11-27T17:09:15.477933Z","shell.execute_reply":"2024-11-27T17:09:15.487293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Genera un reporte de clasificación\nreport = classification_report(y_test, y_pred, target_names=[\"Clase 0\",\"Clase 1\", \"Clase 3\", \"Clase 4\"], output_dict=False)\nprint(\"\\nReporte de clasificación:\")\nprint(report)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:13:57.998895Z","iopub.execute_input":"2024-11-27T17:13:57.999313Z","iopub.status.idle":"2024-11-27T17:13:58.024597Z","shell.execute_reply.started":"2024-11-27T17:13:57.999278Z","shell.execute_reply":"2024-11-27T17:13:58.023521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:14:34.381504Z","iopub.execute_input":"2024-11-27T17:14:34.382647Z","iopub.status.idle":"2024-11-27T17:14:34.621005Z","shell.execute_reply.started":"2024-11-27T17:14:34.382601Z","shell.execute_reply":"2024-11-27T17:14:34.619931Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### En el caso del cumplimiento de objetivos hasta el momento se puede obtener el nivel en el que se ve afectada la persona por el uso del internet o más precisamente del computador, con este nivel se puede obtener los pacientes que necesitan iniciar un plan para mejorar su salud además de ver los pacientes que estan afectados de forma grave.","metadata":{}},{"cell_type":"markdown","source":"### Debido a que el modelo es 'perfecto' teniendo en cuenta el peso de importancia de las caraterísticas obtenidas anteriormente se debe realizar en esta fase una evaluación de las características y su correlación con la variable obetivo que es el nivel de afectación en el paciente","metadata":{}},{"cell_type":"code","source":"df_hitMap = df_process[top_feature_names]\ndf_hitMap['sii'] = df_process['sii']\n\ndf_hitMap\n\ncorrelation_matrix = df_hitMap.corr()\n\n# Crear el heatmap\nplt.figure(figsize=(8, 6))\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', fmt=\".2f\")\nplt.title(\"Heatmap de Correlación\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:49:10.345921Z","iopub.execute_input":"2024-11-27T17:49:10.346731Z","iopub.status.idle":"2024-11-27T17:49:10.925424Z","shell.execute_reply.started":"2024-11-27T17:49:10.346686Z","shell.execute_reply":"2024-11-27T17:49:10.924427Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Se puede observar una fuerte relación con **PCIAT-PCIAT_Total**, esta caracterpistica se deja en fase 3 debido a que se toma como el total de otras columnas esto para reducir las caracteristicas innecesarias, en este caso procedemos a realizar una prueba del modelo de xgboost sin esta caracteristica para ver como afecta en la predicciones.","metadata":{}},{"cell_type":"code","source":"# Separar las características y la variable objetivo\ndf_model = df_process.drop(columns=['sii', 'PCIAT-PCIAT_Total'])\ntarget = df_process['sii']\n\n# Dividir el conjunto de datos en entrenamiento y prueba\nX_train, X_test, y_train, y_test = train_test_split(df_model, target, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:59:02.645735Z","iopub.execute_input":"2024-11-27T17:59:02.646132Z","iopub.status.idle":"2024-11-27T17:59:02.662923Z","shell.execute_reply.started":"2024-11-27T17:59:02.646101Z","shell.execute_reply":"2024-11-27T17:59:02.661985Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dtrain = xgb.DMatrix(X_train, label=y_train)\ndtest = xgb.DMatrix(X_test, label=y_test)\ninitial_params = {\n    'objective': 'multi:softmax',\n    'num_class': len(np.unique(y_train)),\n    'eval_metric': 'merror',\n    'eta': 0.1,  # Tasa de aprendizaje más alta\n    'max_depth': 3,  # Árboles más pequeños\n    'subsample': 0.8,  # Usar solo el 80% de los datos por cada árbol\n    'colsample_bytree': 0.8,  # Usar solo el 80% de las características\n    'min_child_weight': 1,  # Menos regularización\n    'gamma': 0.1,\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:59:05.247804Z","iopub.execute_input":"2024-11-27T17:59:05.248763Z","iopub.status.idle":"2024-11-27T17:59:05.278137Z","shell.execute_reply.started":"2024-11-27T17:59:05.248723Z","shell.execute_reply":"2024-11-27T17:59:05.277339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Crear un modelo básico de XGBoost para búsqueda\nxgb_clf = XGBClassifier(use_label_encoder=False, eval_metric='merror')\n\n# Definir la cuadrícula de búsqueda para mejora de paramatros iniciales\nparam_grid = {\n    'max_depth': [4, 6],\n    'learning_rate': [0.05, .0],\n    'subsample': [0.7, 0.8],\n    'colsample_bytree': [0.7, 0.8],\n    'n_estimators': [50, 100],\n    'gamma': [0, 0.1],\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:59:07.889568Z","iopub.execute_input":"2024-11-27T17:59:07.890420Z","iopub.status.idle":"2024-11-27T17:59:07.897895Z","shell.execute_reply.started":"2024-11-27T17:59:07.890363Z","shell.execute_reply":"2024-11-27T17:59:07.896963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Configurar la búsqueda\ngrid_search = GridSearchCV(estimator=xgb_clf, param_grid=param_grid, scoring='accuracy', cv=3)\ngrid_search.fit(X_train, y_train)\n\n# Obtener los mejores hiperparámetros\nbest_params = grid_search.best_params_\nprint(\"Mejores hiperparámetros encontrados:\", best_params)\n\n# Actualizar los parámetros con los mejores encontrados\nparams = {**initial_params, **best_params}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T17:59:10.680145Z","iopub.execute_input":"2024-11-27T17:59:10.680963Z","iopub.status.idle":"2024-11-27T18:02:32.169083Z","shell.execute_reply.started":"2024-11-27T17:59:10.680922Z","shell.execute_reply":"2024-11-27T18:02:32.168275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Configurar los conjuntos de evaluación\nevals = [(dtrain, 'train'), (dtest, 'eval')]\n\n# Entrenar el modelo con early stopping\nxgb_model = xgb.train(\n    params,\n    dtrain,\n    num_boost_round=1000,\n    evals=evals,\n    early_stopping_rounds=50,\n    verbose_eval=10  # Mostrar métricas cada 10 iteraciones\n)\n\n# Obtener el mejor número de iteraciones\nprint(f\"Mejor número de iteraciones: {xgb_model.best_iteration}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T18:12:18.702597Z","iopub.execute_input":"2024-11-27T18:12:18.703608Z","iopub.status.idle":"2024-11-27T18:12:19.999869Z","shell.execute_reply.started":"2024-11-27T18:12:18.703566Z","shell.execute_reply":"2024-11-27T18:12:19.998652Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Obtener las importancias de características directamente del modelo\nimportance = xgb_model.get_score(importance_type='weight')\n\n# Ordenar las características por importancia y seleccionar las top 10\ntop_features = sorted(importance.items(), key=lambda x: x[1], reverse=True)[:10]\n\n# Extraer solo los nombres de las características\ntop_feature_names = [feature[0] for feature in top_features]\n\n# Mostrar la lista de nombres\nprint(\"Top 10 características importantes:\", top_feature_names)\n\n# Visualizar la importancia (opcional)\nxgb.plot_importance(xgb_model, importance_type='weight', max_num_features=10)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T18:12:23.160468Z","iopub.execute_input":"2024-11-27T18:12:23.160875Z","iopub.status.idle":"2024-11-27T18:12:23.446937Z","shell.execute_reply.started":"2024-11-27T18:12:23.160839Z","shell.execute_reply":"2024-11-27T18:12:23.445896Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predicciones en el conjunto de prueba\ny_pred = xgb_model.predict(dtest)\naccuracy = accuracy_score(y_test, y_pred)\nprint(f\"Precisión del modelo: {accuracy:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T18:12:30.100752Z","iopub.execute_input":"2024-11-27T18:12:30.101434Z","iopub.status.idle":"2024-11-27T18:12:30.112941Z","shell.execute_reply.started":"2024-11-27T18:12:30.101397Z","shell.execute_reply":"2024-11-27T18:12:30.111127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Genera un reporte de clasificación\nreport = classification_report(y_test, y_pred, target_names=[\"Clase 0\",\"Clase 1\", \"Clase 3\", \"Clase 4\"], output_dict=False)\nprint(\"\\nReporte de clasificación:\")\nprint(report)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T18:12:37.383958Z","iopub.execute_input":"2024-11-27T18:12:37.384372Z","iopub.status.idle":"2024-11-27T18:12:37.401155Z","shell.execute_reply.started":"2024-11-27T18:12:37.384323Z","shell.execute_reply":"2024-11-27T18:12:37.400197Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calcular la matriz de confusión\nconf_matrix = confusion_matrix(y_test, y_pred)\n\n# Visualizar la matriz de confusión usando un heatmap de seaborn\nplt.figure(figsize=(8, 6))\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix)\ndisp.plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-27T18:12:44.313879Z","iopub.execute_input":"2024-11-27T18:12:44.314557Z","iopub.status.idle":"2024-11-27T18:12:44.523913Z","shell.execute_reply.started":"2024-11-27T18:12:44.314521Z","shell.execute_reply":"2024-11-27T18:12:44.522742Z"}},"outputs":[],"execution_count":null}]}