{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-06T04:09:45.436578Z","iopub.execute_input":"2021-10-06T04:09:45.437033Z","iopub.status.idle":"2021-10-06T04:09:45.449395Z","shell.execute_reply.started":"2021-10-06T04:09:45.437001Z","shell.execute_reply":"2021-10-06T04:09:45.448219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import logging \nimport time \nimport warnings \nimport catboost as cb \nimport datatable as dt \nimport joblib \nimport lightgbm as lgbm \nimport matplotlib.pyplot as plt \nimport numpy as np \nimport optuna \nimport pandas as pd \nimport seaborn as sns \nimport xgboost as xgb \nfrom optuna.samplers import TPESampler \nfrom sklearn.compose import ( ColumnTransformer, make_column_selector, make_column_transformer, ) \nfrom sklearn.impute import SimpleImputer \nfrom sklearn.metrics import log_loss, mean_squared_error \nfrom sklearn.model_selection import ( KFold, StratifiedKFold, cross_validate, train_test_split, ) \nfrom sklearn.pipeline import Pipeline, make_pipeline \nfrom sklearn.preprocessing import OneHotEncoder, OrdinalEncoder \n\nlogging.basicConfig( format=\"%(asctime)s - %(message)s\", datefmt=\"%d-%b-%y %H:%M:%S\", level=logging.INFO ) \noptuna.logging.set_verbosity(optuna.logging.WARNING) \nwarnings.filterwarnings(\"ignore\") \npd.set_option(\"float_format\", \"{:.5f}\".format)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T04:09:45.451341Z","iopub.execute_input":"2021-10-06T04:09:45.452245Z","iopub.status.idle":"2021-10-06T04:09:45.462384Z","shell.execute_reply.started":"2021-10-06T04:09:45.452206Z","shell.execute_reply":"2021-10-06T04:09:45.461673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"read, scale, show","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"../input/nfl-big-data-bowl-2022/plays.csv\")\ndata.head(5)\nX, y = data.drop(\"playResult\", axis=1), data[[\"playResult\"]].values.flatten()\ny[:10]\nX= pd.DataFrame (X)\n#data.columns","metadata":{"execution":{"iopub.status.busy":"2021-10-06T04:09:45.463694Z","iopub.execute_input":"2021-10-06T04:09:45.464326Z","iopub.status.idle":"2021-10-06T04:09:45.571891Z","shell.execute_reply.started":"2021-10-06T04:09:45.464294Z","shell.execute_reply":"2021-10-06T04:09:45.570761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_cols= [cname for cname in X.columns if X[cname].nunique()<10 and X[cname].dtype==\"object\"]\nnum_cols= [cname for cname in X.columns if X[cname].dtype in [\"int64\",\"float32\",\"float64\"]]\n\nmy_cols = cat_cols + num_cols \nX = X[my_cols].copy()\n\nnumerical_transformer = SimpleImputer(strategy=\"constant\")\ncategorical_transformer = Pipeline(steps =[\n   (\"imput\",SimpleImputer (strategy=\"most_frequent\")),\n   (\"onehot\",  OneHotEncoder(handle_unknown=\"ignore\"))])\n\npreprocessor = ColumnTransformer (transformers=[(\"num\", numerical_transformer,  num_cols), \n                                               (\"cat\", categorical_transformer,  cat_cols)])\nX=preprocessor.fit_transform (X.copy())\n#X=pipe.fit_transform(X.copy())","metadata":{"execution":{"iopub.status.busy":"2021-10-06T04:09:45.57376Z","iopub.execute_input":"2021-10-06T04:09:45.57404Z","iopub.status.idle":"2021-10-06T04:09:45.74127Z","shell.execute_reply.started":"2021-10-06T04:09:45.57401Z","shell.execute_reply":"2021-10-06T04:09:45.740601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install umap-learn \nimport umap\nmanifold = umap.UMAP().fit(X,y)\nX_reduced = manifold.transform(X)\nplt.scatter(X_reduced[:,0], X_reduced[:,1])","metadata":{"execution":{"iopub.status.busy":"2021-10-06T04:09:45.742197Z","iopub.execute_input":"2021-10-06T04:09:45.742939Z","iopub.status.idle":"2021-10-06T04:10:00.523585Z","shell.execute_reply.started":"2021-10-06T04:09:45.742905Z","shell.execute_reply":"2021-10-06T04:10:00.522486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import QuantileTransformer\npipe = make_pipeline (SimpleImputer(strategy=\"mean\"),\n                     QuantileTransformer())\nX = pipe.fit_transform (X.copy())\nmanifold = umap.UMAP().fit(X,y)\nX_reduced1 = manifold.transform(X)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T04:10:00.525338Z","iopub.execute_input":"2021-10-06T04:10:00.525882Z","iopub.status.idle":"2021-10-06T04:10:17.531597Z","shell.execute_reply.started":"2021-10-06T04:10:00.525834Z","shell.execute_reply":"2021-10-06T04:10:17.530453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.scatter (X_reduced1[:,0], X_reduced1[:,1], c=y, s=0.5)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T04:10:17.53301Z","iopub.execute_input":"2021-10-06T04:10:17.53325Z","iopub.status.idle":"2021-10-06T04:10:18.228924Z","shell.execute_reply.started":"2021-10-06T04:10:17.533224Z","shell.execute_reply":"2021-10-06T04:10:18.227916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer\npipe = make_pipeline (PowerTransformer())\nX = pipe.fit_transform (X.copy())\ny_encoded= pd.factorize(y)[0]\nmanifold = umap.UMAP().fit(X, y_encoded)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T04:10:18.23015Z","iopub.execute_input":"2021-10-06T04:10:18.230379Z","iopub.status.idle":"2021-10-06T04:10:35.656887Z","shell.execute_reply.started":"2021-10-06T04:10:18.230353Z","shell.execute_reply":"2021-10-06T04:10:35.655963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#!pip install umap-learn[plot]\nimport umap.plot\n#help(umap.plot.points)\nplt.show()\numap.plot.points(manifold, labels= y, theme=\"fire\")\n#umap.plot.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-06T04:13:52.212521Z","iopub.execute_input":"2021-10-06T04:13:52.213432Z","iopub.status.idle":"2021-10-06T04:13:55.462962Z","shell.execute_reply.started":"2021-10-06T04:13:52.213388Z","shell.execute_reply":"2021-10-06T04:13:55.462026Z"},"trusted":true},"execution_count":null,"outputs":[]}]}