{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt # Me ayudará con las instrucciones necesarias para hacer gráficos\n\nfrom matplotlib.colors import ListedColormap # Me cpermite dar colores a las gráficas y mostrar los datos de forma clara\n# A partir de esta línea, los gráficos de matplotlib aparecerán directamente después de donde se llame la función plot.\n%matplotlib inline \n\ncm = plt.cm.RdBu # Selección del mapa de color \ncm_bright = ListedColormap(['#FF0000', '#0000FF']) ## Selección del mapa de color \n\nimport warnings # Evitar los warnings \nwarnings.filterwarnings('ignore')\n\nimport graphviz # Visualización de los árboles de manera gráfica e incluye el cálculo del GINI\n\nfrom sklearn.model_selection import train_test_split ## Para obtener la muestra para train y para split\n\nimport category_encoders as ce\n\nfrom sklearn import preprocessing ## Funciones para manipular vectores para representarlos en formas más fáciles de interpretar.\n\nimport seaborn as sns ## Para hacer la matriz de correlación\n\nfrom sklearn.model_selection import GridSearchCV ## Para controlar el Cross Validation\nfrom sklearn.linear_model import Lasso ## Para descartar variables y hacer la regresión por Lasso\n\nfrom sklearn.feature_selection import f_regression, mutual_info_regression ## para hacer pruebas de filtrado\n\nfrom sklearn.feature_selection import f_classif, mutual_info_classif ## para clasificar variables (clasificarlas)\n\nfrom sklearn.metrics import mean_squared_error ## Evaluar el MSE y obtener los coeficientes del modelo lineal\n\nfrom sklearn.model_selection import StratifiedKFold ## Para hacer la validación cruzada \n\nfrom sklearn.neighbors import KNeighborsClassifier ## Para aplicar KNN\n\nfrom sklearn.linear_model import LinearRegression ## Para hacer la regresión lineal \n\nfrom sklearn.preprocessing import PolynomialFeatures ## Oara manejar varias variables \n\nfrom sklearn.tree import DecisionTreeClassifier ## para Clasificar usando Random Tree \nfrom sklearn import tree\nfrom sklearn.tree import DecisionTreeRegressor\n\nfrom sklearn.model_selection import KFold ## Para clasificar las variables\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import RandomForestRegressor\nimport io\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-28T11:35:39.422643Z","iopub.execute_input":"2023-01-28T11:35:39.423143Z","iopub.status.idle":"2023-01-28T11:35:39.454921Z","shell.execute_reply.started":"2023-01-28T11:35:39.423106Z","shell.execute_reply":"2023-01-28T11:35:39.453067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_table('/kaggle/input/icecube-neutrinos-in-deep-ice/sensor_geometry.csv',sep=';', decimal='.', on_bad_lines='skip')\nprint(\"This is the shape of the dataframe --> \", data.shape)\nprint(\"Let's see the first 5 lines --> \\n\", data.head())","metadata":{"execution":{"iopub.status.busy":"2023-01-28T11:33:01.009329Z","iopub.execute_input":"2023-01-28T11:33:01.009801Z","iopub.status.idle":"2023-01-28T11:33:01.029668Z","shell.execute_reply.started":"2023-01-28T11:33:01.009766Z","shell.execute_reply":"2023-01-28T11:33:01.028088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"/kaggle/input/icecube-neutrinos-in-deep-ice/train/batch_563.parquet\", \"rb\") as f:\n    data_parquet = f.read()\n\ndf_parquet = pd.read_parquet(\"/kaggle/input/icecube-neutrinos-in-deep-ice/train/batch_563.parquet\")\nprint(\"Let's check what's in the parquet files! --> \", df_parquet)","metadata":{"execution":{"iopub.status.busy":"2023-01-28T11:40:27.545298Z","iopub.execute_input":"2023-01-28T11:40:27.545737Z","iopub.status.idle":"2023-01-28T11:40:29.456144Z","shell.execute_reply.started":"2023-01-28T11:40:27.545703Z","shell.execute_reply":"2023-01-28T11:40:29.454460Z"},"trusted":true},"execution_count":null,"outputs":[]}]}