{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Display only the messages with ERROR, CRITICAL log levels\n!pip install tensorflow_decision_forests -U -qq","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Python packages\nimport os\nimport numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport tensorflow_decision_forests as tfdf\nprint(\"TensorFlow Decision Forests v\" + tfdf.__version__)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define helper functions for plotting training evaluation curves\n\ndef plot_tfdf_model_training_curves(model):\n    # This function was adapted from the following tutorial:\n    # https://www.tensorflow.org/decision_forests/tutorials/beginner_colab\n    logs = model.make_inspector().training_logs()\n    plt.figure(figsize=(12, 4))\n    plt.subplot(1, 2, 1)\n    # Plot accuracy vs number of trees\n    plt.plot([log.num_trees for log in logs], [log.evaluation.accuracy for log in logs])\n    plt.xlabel(\"Number of trees\")\n    plt.ylabel(\"Accuracy (out-of-bag)\")\n    plt.subplot(1, 2, 2)\n    # Plot loss vs number of trees\n    plt.plot([log.num_trees for log in logs], [log.evaluation.loss for log in logs])\n    plt.xlabel(\"Number of trees\")\n    plt.ylabel(\"Logloss (out-of-bag)\")\n    plt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print list of all data and files attached to this notebook\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load to pandas dataframe (for data exploration)\ntrain_df = pd.read_csv('/kaggle/input/stanford-sentiment-treebank/train.csv')\ntest_df = pd.read_csv('/kaggle/input/stanford-sentiment-treebank/test.csv')\n\n# load to tensorflow dataset (for model training)\ntrain_tfds = tfdf.keras.pd_dataframe_to_tf_dataset(train_df, label=\"target\")\ntest_tfds = tfdf.keras.pd_dataframe_to_tf_dataset(test_df)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview first few rows of data\ntrain_df.head(10)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Specify the model.\nmodel_1 = tfdf.keras.GradientBoostedTreesModel(hyperparameter_template=\"benchmark_rank1\")\n\n# Train the model.\nmodel_1.compile(metrics=[tf.keras.metrics.AUC()])\nmodel_1.fit(x=train_tfds)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_tfdf_model_training_curves(model_1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inspector = model_1.make_inspector()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Model type:\", inspector.model_type())\nprint(\"Objective:\", inspector.objective())\nprint(\"Evaluation:\", inspector.evaluation())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df = pd.read_csv('/kaggle/input/stanford-sentiment-treebank/sample_submission.csv')\nsample_submission_df['target'] = model_1.predict(test_tfds)\nsample_submission_df.to_csv('/kaggle/working/submission_raw.csv', index=False)\nsample_submission_df.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade tensorflow-hub","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow_hub as hub\n# NNLM (https://tfhub.dev/google/nnlm-en-dim128/2) is also a good choice.\nhub_url = \"http://tfhub.dev/google/universal-sentence-encoder/4\"\nembedding = hub.KerasLayer(hub_url)\n\nsentence = tf.keras.layers.Input(shape=(), name=\"sentence\", dtype=tf.string)\nembedded_sentence = embedding(sentence)\n\nraw_inputs = {\"sentence\": sentence}\nprocessed_inputs = {\"embedded_sentence\": embedded_sentence}\npreprocessor = tf.keras.Model(inputs=raw_inputs, outputs=processed_inputs)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Specify the model.\nmodel_2 = tfdf.keras.GradientBoostedTreesModel(hyperparameter_template=\"benchmark_rank1\", preprocessing=preprocessor)\n\n# Train the model.\nmodel_2.compile(metrics=[tf.keras.metrics.AUC()])\nmodel_2.fit(x=train_tfds)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_tfdf_model_training_curves(model_2)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inspector = model_2.make_inspector()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Model type:\", inspector.model_type())\nprint(\"Objective:\", inspector.objective())\nprint(\"Evaluation:\", inspector.evaluation())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df = pd.read_csv('/kaggle/input/stanford-sentiment-treebank/sample_submission.csv')\nsample_submission_df['target'] = model_2.predict(test_tfds)\nsample_submission_df.to_csv('/kaggle/working/submission.csv', index=False)\nsample_submission_df.head()","metadata":{},"execution_count":null,"outputs":[]}]}