{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T16:07:49.750037Z","iopub.execute_input":"2022-08-12T16:07:49.750515Z","iopub.status.idle":"2022-08-12T16:07:49.761989Z","shell.execute_reply.started":"2022-08-12T16:07:49.750473Z","shell.execute_reply":"2022-08-12T16:07:49.760803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Using skops to host your models on Hugging Face Hub\nThis notebook shows you how you can use [skops](https://skops.readthedocs.io/) to improve your data science workflows with scikit-learn. We will have end-to-end example for Kaggle Tabular Playground Series of August 2022.","metadata":{}},{"cell_type":"markdown","source":"## Install skops","metadata":{}},{"cell_type":"code","source":"#!pip install skops","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:42:20.000537Z","iopub.execute_input":"2022-08-12T16:42:20.000960Z","iopub.status.idle":"2022-08-12T16:42:20.005212Z","shell.execute_reply.started":"2022-08-12T16:42:20.000926Z","shell.execute_reply":"2022-08-12T16:42:20.004298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import libraries","metadata":{}},{"cell_type":"code","source":"import skops\nimport sklearn\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.273144Z","iopub.execute_input":"2022-08-12T16:08:01.273524Z","iopub.status.idle":"2022-08-12T16:08:01.279217Z","shell.execute_reply.started":"2022-08-12T16:08:01.273487Z","shell.execute_reply":"2022-08-12T16:08:01.277670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's take a look at the dataset\nTarget variable is a binary category. We have couple of numerical and categorical variables.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.280555Z","iopub.execute_input":"2022-08-12T16:08:01.280918Z","iopub.status.idle":"2022-08-12T16:08:01.433127Z","shell.execute_reply.started":"2022-08-12T16:08:01.280882Z","shell.execute_reply":"2022-08-12T16:08:01.431902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"failure\"].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.436722Z","iopub.execute_input":"2022-08-12T16:08:01.437150Z","iopub.status.idle":"2022-08-12T16:08:01.445258Z","shell.execute_reply.started":"2022-08-12T16:08:01.437117Z","shell.execute_reply":"2022-08-12T16:08:01.444066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Encode categorical variables, impute missing values\nWe will impute mean for the numerical attribues and measurements. ","metadata":{}},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.447099Z","iopub.execute_input":"2022-08-12T16:08:01.447438Z","iopub.status.idle":"2022-08-12T16:08:01.558557Z","shell.execute_reply.started":"2022-08-12T16:08:01.447409Z","shell.execute_reply":"2022-08-12T16:08:01.557437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Take a look at the missing values and data types.","metadata":{}},{"cell_type":"code","source":"df.isna().any()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.560497Z","iopub.execute_input":"2022-08-12T16:08:01.560849Z","iopub.status.idle":"2022-08-12T16:08:01.573857Z","shell.execute_reply.started":"2022-08-12T16:08:01.560796Z","shell.execute_reply":"2022-08-12T16:08:01.572810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.575878Z","iopub.execute_input":"2022-08-12T16:08:01.576226Z","iopub.status.idle":"2022-08-12T16:08:01.585351Z","shell.execute_reply.started":"2022-08-12T16:08:01.576194Z","shell.execute_reply":"2022-08-12T16:08:01.584190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see the cardinality of categorical variables.","metadata":{}},{"cell_type":"code","source":"print(df.product_code.nunique())\nprint(df.attribute_0.nunique())\nprint(df.attribute_0.nunique())\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.586538Z","iopub.execute_input":"2022-08-12T16:08:01.587124Z","iopub.status.idle":"2022-08-12T16:08:01.602366Z","shell.execute_reply.started":"2022-08-12T16:08:01.587085Z","shell.execute_reply":"2022-08-12T16:08:01.601468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing \nWe will use OneHotEncoder to encode our categorical variables, SimpleImputer to impute missing values and put them all in a ColumnTransformer. We will then use the transformer in our machine learning pipeline to have an end-to-end object for better reproducibility.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\n\ncolumn_transformer_pipeline = ColumnTransformer([\n                (\"loading_missing_value_imputer\", SimpleImputer(strategy=\"mean\"), [\"loading\"]),\n                (\"numerical_missing_value_imputer\", SimpleImputer(strategy=\"mean\"), list(df.columns[df.dtypes == 'float64'])),\n                (\"attribute_0_encoder\", OneHotEncoder(categories = \"auto\"), [\"attribute_0\"]),\n                (\"attribute_1_encoder\", OneHotEncoder(categories = \"auto\"), [\"attribute_1\"]),\n                (\"product_code_encoder\", OneHotEncoder(categories = \"auto\"), [\"product_code\"])])","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.603692Z","iopub.execute_input":"2022-08-12T16:08:01.604304Z","iopub.status.idle":"2022-08-12T16:08:01.612756Z","shell.execute_reply.started":"2022-08-12T16:08:01.604268Z","shell.execute_reply":"2022-08-12T16:08:01.611678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop([\"id\"], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.616244Z","iopub.execute_input":"2022-08-12T16:08:01.616897Z","iopub.status.idle":"2022-08-12T16:08:01.628662Z","shell.execute_reply.started":"2022-08-12T16:08:01.616855Z","shell.execute_reply":"2022-08-12T16:08:01.627386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\nfrom sklearn.pipeline import Pipeline\npipeline = Pipeline([\n    ('transformation', column_transformer_pipeline),\n    ('model', DecisionTreeClassifier(max_depth=4))\n])","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.630096Z","iopub.execute_input":"2022-08-12T16:08:01.631034Z","iopub.status.idle":"2022-08-12T16:08:01.640519Z","shell.execute_reply.started":"2022-08-12T16:08:01.630995Z","shell.execute_reply":"2022-08-12T16:08:01.639257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.drop([\"failure\"], axis = 1)\ny = df.failure","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.642270Z","iopub.execute_input":"2022-08-12T16:08:01.643448Z","iopub.status.idle":"2022-08-12T16:08:01.656699Z","shell.execute_reply.started":"2022-08-12T16:08:01.643404Z","shell.execute_reply":"2022-08-12T16:08:01.655346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.658597Z","iopub.execute_input":"2022-08-12T16:08:01.659907Z","iopub.status.idle":"2022-08-12T16:08:01.680523Z","shell.execute_reply.started":"2022-08-12T16:08:01.659853Z","shell.execute_reply":"2022-08-12T16:08:01.679150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipeline.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.682078Z","iopub.execute_input":"2022-08-12T16:08:01.682574Z","iopub.status.idle":"2022-08-12T16:08:01.927531Z","shell.execute_reply.started":"2022-08-12T16:08:01.682526Z","shell.execute_reply":"2022-08-12T16:08:01.926319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = pipeline.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.929273Z","iopub.execute_input":"2022-08-12T16:08:01.929842Z","iopub.status.idle":"2022-08-12T16:08:01.956267Z","shell.execute_reply.started":"2022-08-12T16:08:01.929778Z","shell.execute_reply":"2022-08-12T16:08:01.955125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We will now save the model and create a model card with metrics about our model!","metadata":{}},{"cell_type":"markdown","source":"We will use `hub_utils` for model hosting and `card` to create a model card. First, we will initialize a local repository to contain our model, model configuration, model card and anything else that we want. (e.g. plots)","metadata":{}},{"cell_type":"code","source":"from skops import card, hub_utils\nimport pickle\n\nmodel_path = \"model.pkl\"\nlocal_repo = \"decision-tree-playground-kaggle\"\n\nwith open(model_path, mode=\"bw\") as f:\n    pickle.dump(pipeline, file=f)\n\nhub_utils.init(\nmodel=model_path, \nrequirements=[f\"scikit-learn={sklearn.__version__}\"], \ndst=local_repo,\ntask=\"tabular-classification\",\ndata=X_test,\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.957800Z","iopub.execute_input":"2022-08-12T16:08:01.958544Z","iopub.status.idle":"2022-08-12T16:08:01.971908Z","shell.execute_reply.started":"2022-08-12T16:08:01.958496Z","shell.execute_reply":"2022-08-12T16:08:01.970902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## We will now create our card 🃏 ","metadata":{}},{"cell_type":"markdown","source":"Creating the model card is as simple as instantiating `Card` class of `skops`. Calling `metadata_from_config` method will create metadata section of the model card from configuration file. We will use `add` method to pass information to our model card.","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\nmodel_card = card.Card(pipeline, metadata=card.metadata_from_config(Path(local_repo)))\n\n## let's fill some information about the model\nlimitations = \"This model is not ready to be used in production.\"\nmodel_description = \"This is a DecisionTreeClassifier model built for Kaggle Tabular Playground Series August 2022, trained on supersoaker production failures dataset.\"\nmodel_card_authors = \"huggingface\"\nget_started_code = f\"import pickle \\nwith open({local_repo}/{model_path}, 'rb') as file: \\n    clf = pickle.load(file)\"\n\n# pass this information to the card\nmodel_card.add(\n    get_started_code=get_started_code,\n    model_card_authors=model_card_authors,\n    limitations=limitations,\n    model_description=model_description,\n)\n# adding methods return the model card itself for easy method chaining","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:01.973796Z","iopub.execute_input":"2022-08-12T16:08:01.974696Z","iopub.status.idle":"2022-08-12T16:08:02.071532Z","shell.execute_reply.started":"2022-08-12T16:08:01.974655Z","shell.execute_reply":"2022-08-12T16:08:02.070310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will now plot and create insights about our model and write them to the model card. \nPipeline includes the decision tree in the last step of it, you can see the content of pipeline as a tuple. The second element of the tuple includes the object -the tree model- itself so if we want to plot the tree we have to first get it from the pipeline. (see below)","metadata":{}},{"cell_type":"code","source":"pipeline.steps[-1][1]","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:02.073020Z","iopub.execute_input":"2022-08-12T16:08:02.073400Z","iopub.status.idle":"2022-08-12T16:08:02.082681Z","shell.execute_reply.started":"2022-08-12T16:08:02.073367Z","shell.execute_reply":"2022-08-12T16:08:02.080924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can use `add_metrics` to pass metrics to our model card, which skops will parse into a table for us. We will use `add_plots` to add our plots. ","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, f1_score, ConfusionMatrixDisplay, confusion_matrix\nmodel_card.add(eval_method=\"The model is evaluated using test split, on accuracy and F1 score with micro average.\")\nmodel_card.add_metrics(accuracy=accuracy_score(y_test, y_pred))\nmodel_card.add_metrics(**{\"f1 score\": f1_score(y_test, y_pred, average=\"micro\")})\n\nmodel = pipeline.steps[-1][1]\n# we will plot the tree and add the plot to our card\nfrom sklearn.tree import plot_tree\nplt.figure()\nplot_tree(model,filled=True)  \nplt.savefig(f'{local_repo}/tree.png',format='png',bbox_inches = \"tight\")\n\n# let's make a prediction and evaluate the model\n\ny_pred = pipeline.predict(X_test)\ncm = confusion_matrix(y_test, y_pred, labels=model.classes_)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=model.classes_)\ndisp.plot()\n# save the plot\nplt.savefig(Path(local_repo) / \"confusion_matrix.png\")\n# add figures to model card with their new sections as keys to the dictionary\nmodel_card.add_plot(**{\"Tree Plot\": f'{local_repo}/tree.png', \"Confusion Matrix\": f\"{local_repo}/confusion_matrix.png\"})","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:02.084852Z","iopub.execute_input":"2022-08-12T16:08:02.085287Z","iopub.status.idle":"2022-08-12T16:08:05.482006Z","shell.execute_reply.started":"2022-08-12T16:08:02.085232Z","shell.execute_reply":"2022-08-12T16:08:05.480747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will now save our model card.","metadata":{}},{"cell_type":"code","source":"model_card.save(f\"{local_repo}/README.md\")","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:05.483575Z","iopub.execute_input":"2022-08-12T16:08:05.484066Z","iopub.status.idle":"2022-08-12T16:08:05.518708Z","shell.execute_reply.started":"2022-08-12T16:08:05.484021Z","shell.execute_reply":"2022-08-12T16:08:05.517522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's push our model repository to Hub! \nHugging Face Hub requires us to authenticate ourselves, we can do that using `notebook_login`\n","metadata":{}},{"cell_type":"code","source":"from huggingface_hub import notebook_login\nnotebook_login()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:04:27.699235Z","iopub.execute_input":"2022-08-12T16:04:27.699722Z","iopub.status.idle":"2022-08-12T16:04:27.744734Z","shell.execute_reply.started":"2022-08-12T16:04:27.699676Z","shell.execute_reply":"2022-08-12T16:04:27.743310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can push our model using `hub_utils.push`","metadata":{}},{"cell_type":"code","source":"# if the repository doesn't exist remotely on the Hugging Face Hub, it will be created when we set create_remote to True\nrepo_id = \"scikit-learn/tabular-playground\"\nhub_utils.push(\n    repo_id=repo_id,\n    source=local_repo,\n    token=token,\n    commit_message=\"pushing files to the repo from the example!\",\n    create_remote=True,\n)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-12T16:08:15.078202Z","iopub.execute_input":"2022-08-12T16:08:15.078653Z","iopub.status.idle":"2022-08-12T16:08:18.508240Z","shell.execute_reply.started":"2022-08-12T16:08:15.078614Z","shell.execute_reply":"2022-08-12T16:08:18.506828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## After we push it, the widget is enabled like below:","metadata":{}},{"cell_type":"markdown","source":"![Widget](https://huggingface.co/scikit-learn/tabular-playground/resolve/main/widget_screenshot.png)","metadata":{}},{"cell_type":"markdown","source":"# See how repository and our model card looks like [here](https://huggingface.co/scikit-learn/tabular-playground)  ✨","metadata":{}}]}