{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nfrom itertools import chain\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Análisis de datos exploratorio\n## Caso de estudio: Plant pathology dataset\n\n","metadata":{}},{"cell_type":"markdown","source":"### Cargar datos\n","metadata":{}},{"cell_type":"code","source":"df_datos = pd.read_csv('/kaggle/input/plant-pathology-2021-fgvc8/train.csv')\nnRow, nCol = df_datos.shape\nprint(f'There are {nRow} rows and {nCol} columns')\ndf_datos.sample(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tipos de datos\n","metadata":{}},{"cell_type":"code","source":"col_labels = 'labels'\n\nprint('Tipos de datos: \\n', df_datos.dtypes)\nprint('Tipo de datos de las etiquetas: ', type(df_datos[col_labels][0]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Columnas binarias","metadata":{}},{"cell_type":"code","source":"\n## Here you may want to create some extra columns in your table with binary indicators of certain diseases \n## rather than working directly with the 'Finding Labels' column\n\nall_labels = np.unique(list(chain(*df_datos[col_labels].map(lambda x: x.split(' ')).tolist())))\nall_labels = [x for x in all_labels if len(x)>0]\nprint('All Labels ({}): {}'.format(len(all_labels), all_labels))\n\nfor c_label in all_labels:\n    if len(c_label)>1: # leave out empty labels\n        df_datos[c_label] = df_datos[col_labels].map(lambda finding: 1.0 if c_label in finding else 0)\ndf_datos.sample(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Estudio de la población\n\n### Distribución de los hallazgos","metadata":{}},{"cell_type":"code","source":"def splitClasses(string):\n    return string.split(' ')\n\ndef getCountsDataFrame(df, column, labels):\n    col = df[column]\n    diccionario  = {l:0 for l in labels}\n    for element in col:\n        lsplit = splitClasses(element)\n        #print(lsplit)\n        for l_individual in lsplit:\n            #print(diccionario[l_individual])\n            diccionario[l_individual] += 1\n    return diccionario\n\ndef plotCounts(counts_dict, graphWidth, name='Counts'):\n    plt.figure(num=None, figsize=(graphWidth, graphWidth), dpi=80, facecolor='w', edgecolor='k')\n    plt.xticks(rotation='vertical')\n    plt.title(name)\n    plt.bar(*zip(*counts_dict.items()), color=['red', 'green', 'blue', 'cyan', 'black'])\n    plt.show()\n\n\n# Plot disease labels distribution\ndict_counts = getCountsDataFrame(df_datos, col_labels, all_labels)\n\nplotCounts(dict_counts, 6, name= 'Distribución de las enfermedades')\n\n#print('Neumonia Cases:', dict_counts['Pneumonia'])\n#print('No neumonia Cases:', nRow - dict_counts['Pneumonia'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to get unique values\ndef unique(list1):\n    # intilize a null list\n    unique_list = []\n    # traverse for all elements\n    for x in list1:\n        # check if exists in unique_list or not\n        if x not in unique_list:\n            unique_list.append(x)\n    return unique_list\n\ndef getAllDifferentClasses(dataframe, column):\n    labels = dataframe[column].unique()\n    anidada = [splitClasses(cs) for cs in labels]\n    labels = [l for lista in anidada for l in lista]\n    return unique(labels)\n\ndef coOcurrencia(dataframe, column):\n    #labelsmix = dataframe[column].unique()\n    labels = getAllDifferentClasses(dataframe, column)\n    n_labels = len(labels)\n    matrix_coo = np.zeros((n_labels,n_labels))\n    \n    for i, la in enumerate(labels):\n        max_count = 0\n        normalize = False;\n        for j, lb in enumerate(labels):\n            #search pair (la, lb)\n            count = countPairs(dataframe, column, la, lb)\n            matrix_coo[i][j] = count\n            if count > max_count:\n                max_count = count\n                normalize = True\n        if normalize:\n            matrix_coo[i] = matrix_coo[i] * (1/max_count)\n    return labels, matrix_coo\n\ndef plotCorrelationMatrix(labels, matrix, graphWidth, title='Correlation Matrix for'):\n    plt.figure(num=None, figsize=(graphWidth, graphWidth), dpi=80, facecolor='w', edgecolor='k')\n    corrMat = plt.matshow(matrix, fignum = 1)\n    plt.xticks(range(len(labels)), labels, rotation=90)\n    plt.yticks(range(len(labels)), labels)\n    plt.gca().xaxis.tick_bottom()\n    plt.colorbar(corrMat)\n    plt.title(title, fontsize=15)\n    plt.show()\n    \ndef countPairs(df, column, la, lb):\n    col = df[column]\n    suma = 0\n    for element in col:\n        lsplit = splitClasses(element)\n        if la in lsplit and lb in lsplit:\n            suma += 1\n    return suma","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels, matco = coOcurrencia(df_datos, col_labels)\nplotCorrelationMatrix(labels, matco, 9, title = 'Matriz de correlación')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Conclusiones\n\n- Las clases están desbalanceadas\n- Etiquetas múltiples (multiclase)\n- Solo existe información de la etiqueta","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}