Update code to v2.11.0

32e4ca51 · qianyj · 9485aa1d · 71060f67 · 32e4ca51 · 32e4ca51
Commit 32e4ca51 authored Nov 28, 2023 by qianyj
20 changed files
--- a/official/benchmark/keras_imagenet_benchmark.py
+++ b/official/benchmark/keras_imagenet_benchmark.py
-# Lint as: python3
 # Copyright 2018 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");

--- a/official/benchmark/models/resnet_imagenet_main.py
+++ b/official/benchmark/models/resnet_imagenet_main.py
@@ -74,8 +74,6 @@ def run(flags_obj):
  Returns:
    Dictionary of training and eval stats.
  """
-  keras_utils.set_session_config(
-      enable_xla=flags_obj.enable_xla)
  # Execute flag override logic for better model performance
  if flags_obj.tf_gpu_thread_mode:
    keras_utils.set_gpu_thread_mode_and_count(
@@ -251,7 +249,8 @@ def run(flags_obj):
        optimizer=optimizer,
        metrics=(['sparse_categorical_accuracy']
                 if flags_obj.report_accuracy_metrics else None),
-        run_eagerly=flags_obj.run_eagerly)
+        run_eagerly=flags_obj.run_eagerly,
+        jit_compile=flags_obj.enable_xla)

  train_epochs = flags_obj.train_epochs


--- a/official/benchmark/nhnet_benchmark.py
+++ b/official/benchmark/nhnet_benchmark.py
-# Lint as: python3
 # Copyright 2020 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");

--- a/official/benchmark/resnet50_keras_core.py
+++ b/official/benchmark/resnet50_keras_core.py
-# Lint as: python3
 # Copyright 2020 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
@@ -15,6 +14,7 @@
 # ==============================================================================
 """Resnet50 Keras core benchmark."""

+import statistics
 import tempfile
 import time

@@ -100,7 +100,7 @@ class Resnet50KerasCoreBenchmark(perfzero_benchmark.PerfZeroBenchmark):
    wall_times = []
    for _ in range(num_trials):
      wall_times.append(_run_benchmark())
-    avg_wall_time = sum(wall_times) / float(len(wall_times))
+    avg_wall_time = statistics.mean(wall_times)
    self.report_benchmark(iters=-1, wall_time=avg_wall_time)

  def benchmark_1_gpu_max_3(self):
@@ -111,5 +111,21 @@ class Resnet50KerasCoreBenchmark(perfzero_benchmark.PerfZeroBenchmark):
    max_wall_time = max(wall_times)
    self.report_benchmark(iters=-1, wall_time=max_wall_time)

+  def benchmark_1_gpu_min_3(self):
+    num_trials = 3
+    wall_times = []
+    for _ in range(num_trials):
+      wall_times.append(_run_benchmark())
+    min_wall_time = min(wall_times)
+    self.report_benchmark(iters=-1, wall_time=min_wall_time)
+
+  def benchmark_1_gpu_med_3(self):
+    num_trials = 3
+    wall_times = []
+    for _ in range(num_trials):
+      wall_times.append(_run_benchmark())
+    med_wall_time = statistics.median(wall_times)
+    self.report_benchmark(iters=-1, wall_time=med_wall_time)
+
 if __name__ == "__main__":
  tf.test.main()
--- a/official/benchmark/shakespeare_benchmark.py
+++ b/official/benchmark/shakespeare_benchmark.py
@@ -331,7 +331,7 @@ class ShakespeareKerasBenchmarkReal(ShakespeareBenchmarkBase):
  def benchmark_xla_8_gpu(self):
    """Benchmark 8 gpu w/xla."""
    self._setup()
-    FLAGS.num_gpus = 1
+    FLAGS.num_gpus = 8
    FLAGS.batch_size = 64 * 8
    FLAGS.log_steps = 10
    FLAGS.enable_xla = True

--- a/official/benchmark/xlnet_benchmark.py
+++ b/official/benchmark/xlnet_benchmark.py
-# Copyright 2019 The TensorFlow Authors. All Rights Reserved.
-#
-# Licensed under the Apache License, Version 2.0 (the "License");
-# you may not use this file except in compliance with the License.
-# You may obtain a copy of the License at
-#
-#     http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-# ==============================================================================
-"""Executes XLNet benchmarks and accuracy tests."""
-
-from __future__ import absolute_import
-from __future__ import division
-from __future__ import print_function
-
-import json
-import os
-import time
-
-# pylint: disable=g-bad-import-order
-
-from absl import flags
-from absl.testing import flagsaver
-import tensorflow as tf
-# pylint: enable=g-bad-import-order
-
-from official.benchmark import bert_benchmark_utils as benchmark_utils
-from official.benchmark import owner_utils
-from official.nlp.xlnet import run_classifier
-from official.nlp.xlnet import run_squad
-from official.benchmark import benchmark_wrappers
-
-
-# pylint: disable=line-too-long
-PRETRAINED_CHECKPOINT_PATH = 'gs://cloud-tpu-checkpoints/xlnet/large/xlnet_model-1'
-CLASSIFIER_TRAIN_DATA_PATH = 'gs://tf-perfzero-data/xlnet/imdb/spiece.model.len-512.train.tf_record'
-CLASSIFIER_EVAL_DATA_PATH = 'gs://tf-perfzero-data/xlnet/imdb/spiece.model.len-512.dev.eval.tf_record'
-SQUAD_DATA_PATH = 'gs://tf-perfzero-data/xlnet/squadv2_cased/'
-# pylint: enable=line-too-long
-
-FLAGS = flags.FLAGS
-
-
-class XLNetBenchmarkBase(benchmark_utils.BertBenchmarkBase):
-  """Base class to hold methods common to test classes in the module."""
-
-  def __init__(self, output_dir=None, tpu=None):
-    super(XLNetBenchmarkBase, self).__init__(output_dir=output_dir, tpu=tpu)
-    self.num_epochs = None
-    self.num_steps_per_epoch = None
-
-  @flagsaver.flagsaver
-  def _run_xlnet_classifier(self):
-    """Starts XLNet classification task."""
-    run_classifier.main(unused_argv=None)
-
-  @flagsaver.flagsaver
-  def _run_xlnet_squad(self):
-    """Starts XLNet classification task."""
-    run_squad.main(unused_argv=None)
-
-
-class XLNetClassifyAccuracy(XLNetBenchmarkBase):
-  """Short accuracy test for XLNet classifier model.
-
-  Tests XLNet classification task model accuracy. The naming
-  convention of below test cases follow
-  `benchmark_(number of gpus)_gpu_(dataset type)` format.
-  """
-
-  def __init__(self, output_dir=None, tpu=None, **kwargs):
-    self.train_data_path = CLASSIFIER_TRAIN_DATA_PATH
-    self.eval_data_path = CLASSIFIER_EVAL_DATA_PATH
-    self.pretrained_checkpoint_path = PRETRAINED_CHECKPOINT_PATH
-
-    super(XLNetClassifyAccuracy, self).__init__(output_dir=output_dir, tpu=tpu)
-
-  @benchmark_wrappers.enable_runtime_flags
-  def _run_and_report_benchmark(self,
-                                training_summary_path,
-                                min_accuracy=0.95,
-                                max_accuracy=0.97):
-    """Starts XLNet accuracy benchmark test."""
-
-    start_time_sec = time.time()
-    self._run_xlnet_classifier()
-    wall_time_sec = time.time() - start_time_sec
-
-    with tf.io.gfile.GFile(training_summary_path, 'rb') as reader:
-      summary = json.loads(reader.read().decode('utf-8'))
-
-    super(XLNetClassifyAccuracy, self)._report_benchmark(
-        stats=summary,
-        wall_time_sec=wall_time_sec,
-        min_accuracy=min_accuracy,
-        max_accuracy=max_accuracy)
-
-  def _setup(self):
-    super(XLNetClassifyAccuracy, self)._setup()
-    FLAGS.test_data_size = 25024
-    FLAGS.train_batch_size = 16
-    FLAGS.seq_len = 512
-    FLAGS.mem_len = 0
-    FLAGS.n_layer = 24
-    FLAGS.d_model = 1024
-    FLAGS.d_embed = 1024
-    FLAGS.n_head = 16
-    FLAGS.d_head = 64
-    FLAGS.d_inner = 4096
-    FLAGS.untie_r = True
-    FLAGS.n_class = 2
-    FLAGS.ff_activation = 'gelu'
-    FLAGS.strategy_type = 'mirror'
-    FLAGS.learning_rate = 2e-5
-    FLAGS.train_steps = 4000
-    FLAGS.warmup_steps = 500
-    FLAGS.iterations = 200
-    FLAGS.bi_data = False
-    FLAGS.init_checkpoint = self.pretrained_checkpoint_path
-    FLAGS.train_tfrecord_path = self.train_data_path
-    FLAGS.test_tfrecord_path = self.eval_data_path
-
-  @owner_utils.Owner('tf-model-garden')
-  def benchmark_8_gpu_imdb(self):
-    """Run XLNet model accuracy test with 8 GPUs."""
-    self._setup()
-    FLAGS.model_dir = self._get_model_dir('benchmark_8_gpu_imdb')
-    # Sets timer_callback to None as we do not use it now.
-    self.timer_callback = None
-
-    summary_path = os.path.join(FLAGS.model_dir,
-                                'summaries/training_summary.txt')
-    self._run_and_report_benchmark(summary_path)
-
-  @owner_utils.Owner('tf-model-garden')
-  def benchmark_2x2_tpu_imdb(self):
-    """Run XLNet model accuracy test on 2x2 tpu."""
-    self._setup()
-    FLAGS.strategy_type = 'tpu'
-    FLAGS.model_dir = self._get_model_dir('benchmark_2x2_tpu_imdb')
-    # Sets timer_callback to None as we do not use it now.
-    self.timer_callback = None
-
-    summary_path = os.path.join(FLAGS.model_dir,
-                                'summaries/training_summary.txt')
-    self._run_and_report_benchmark(summary_path)
-
-
-class XLNetSquadAccuracy(XLNetBenchmarkBase):
-  """Short accuracy test for XLNet squad model.
-
-  Tests XLNet squad task model accuracy. The naming
-  convention of below test cases follow
-  `benchmark_(number of gpus)_gpu_(dataset type)` format.
-  """
-
-  def __init__(self, output_dir=None, tpu=None, **kwargs):
-    self.train_data_path = SQUAD_DATA_PATH
-    self.predict_file = os.path.join(SQUAD_DATA_PATH, 'dev-v2.0.json')
-    self.test_data_path = os.path.join(SQUAD_DATA_PATH, '12048.eval.tf_record')
-    self.spiece_model_file = os.path.join(SQUAD_DATA_PATH, 'spiece.cased.model')
-    self.pretrained_checkpoint_path = PRETRAINED_CHECKPOINT_PATH
-
-    super(XLNetSquadAccuracy, self).__init__(output_dir=output_dir, tpu=tpu)
-
-  @benchmark_wrappers.enable_runtime_flags
-  def _run_and_report_benchmark(self,
-                                training_summary_path,
-                                min_accuracy=87.0,
-                                max_accuracy=89.0):
-    """Starts XLNet accuracy benchmark test."""
-
-    start_time_sec = time.time()
-    self._run_xlnet_squad()
-    wall_time_sec = time.time() - start_time_sec
-
-    with tf.io.gfile.GFile(training_summary_path, 'rb') as reader:
-      summary = json.loads(reader.read().decode('utf-8'))
-
-    super(XLNetSquadAccuracy, self)._report_benchmark(
-        stats=summary,
-        wall_time_sec=wall_time_sec,
-        min_accuracy=min_accuracy,
-        max_accuracy=max_accuracy)
-
-  def _setup(self):
-    super(XLNetSquadAccuracy, self)._setup()
-    FLAGS.train_batch_size = 16
-    FLAGS.seq_len = 512
-    FLAGS.mem_len = 0
-    FLAGS.n_layer = 24
-    FLAGS.d_model = 1024
-    FLAGS.d_embed = 1024
-    FLAGS.n_head = 16
-    FLAGS.d_head = 64
-    FLAGS.d_inner = 4096
-    FLAGS.untie_r = True
-    FLAGS.ff_activation = 'gelu'
-    FLAGS.strategy_type = 'mirror'
-    FLAGS.learning_rate = 3e-5
-    FLAGS.train_steps = 8000
-    FLAGS.warmup_steps = 1000
-    FLAGS.iterations = 1000
-    FLAGS.bi_data = False
-    FLAGS.init_checkpoint = self.pretrained_checkpoint_path
-    FLAGS.train_tfrecord_path = self.train_data_path
-    FLAGS.test_tfrecord_path = self.test_data_path
-    FLAGS.spiece_model_file = self.spiece_model_file
-    FLAGS.predict_file = self.predict_file
-    FLAGS.adam_epsilon = 1e-6
-    FLAGS.lr_layer_decay_rate = 0.75
-
-  @owner_utils.Owner('tf-model-garden')
-  def benchmark_8_gpu_squadv2(self):
-    """Run XLNet model squad v2 accuracy test with 8 GPUs."""
-    self._setup()
-    FLAGS.model_dir = self._get_model_dir('benchmark_8_gpu_squadv2')
-    FLAGS.predict_dir = FLAGS.model_dir
-    # Sets timer_callback to None as we do not use it now.
-    self.timer_callback = None
-
-    summary_path = os.path.join(FLAGS.model_dir,
-                                'summaries/training_summary.txt')
-    self._run_and_report_benchmark(summary_path)
-
-  @owner_utils.Owner('tf-model-garden')
-  def benchmark_2x2_tpu_squadv2(self):
-    """Run XLNet model squad v2 accuracy test on 2x2 tpu."""
-    self._setup()
-    FLAGS.strategy_type = 'tpu'
-    FLAGS.model_dir = self._get_model_dir('benchmark_2x2_tpu_squadv2')
-    FLAGS.predict_dir = FLAGS.model_dir
-    # Sets timer_callback to None as we do not use it now.
-    self.timer_callback = None
-
-    summary_path = os.path.join(FLAGS.model_dir,
-                                'summaries/training_summary.txt')
-    self._run_and_report_benchmark(summary_path)
-
-
-if __name__ == '__main__':
-  tf.test.main()
--- a/official/colab/README.md
+++ b/official/colab/README.md
+# Moved
+
+These files have moved to:
+https://github.com/tensorflow/models/blob/master/docs
\ No newline at end of file
--- a/official/colab/decoding_api_in_tf_nlp.ipynb
+++ b/official/colab/decoding_api_in_tf_nlp.ipynb
-{
-  "cells": [
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "vXLA5InzXydn"
-      },
-      "source": [
-        "##### Copyright 2021 The TensorFlow Authors."
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "cellView": "form",
-        "id": "RuRlpLL-X0R_"
-      },
-      "outputs": [],
-      "source": [
-        "#@title Licensed under the Apache License, Version 2.0 (the \"License\");\n",
-        "# you may not use this file except in compliance with the License.\n",
-        "# You may obtain a copy of the License at\n",
-        "#\n",
-        "# https://www.apache.org/licenses/LICENSE-2.0\n",
-        "#\n",
-        "# Unless required by applicable law or agreed to in writing, software\n",
-        "# distributed under the License is distributed on an \"AS IS\" BASIS,\n",
-        "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
-        "# See the License for the specific language governing permissions and\n",
-        "# limitations under the License."
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "fsACVQpVSifi"
-      },
-      "source": [
-        "### Install the TensorFlow Model Garden pip package\n",
-        "\n",
-        "*  `tf-models-official` is the stable Model Garden package. Note that it may not include the latest changes in the `tensorflow_models` github repo. To include latest changes, you may install `tf-models-nightly`,\n",
-        "which is the nightly Model Garden package created daily automatically.\n",
-        "*  pip will install all models and dependencies automatically."
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "hYEwGTeCXnnX"
-      },
-      "source": [
-        "\u003ctable class=\"tfo-notebook-buttons\" align=\"left\"\u003e\n",
-        "  \u003ctd\u003e\n",
-        "    \u003ca target=\"_blank\" href=\"https://www.tensorflow.org/official_models/tutorials/decoding_api_in_tf_nlp.ipynb\"\u003e\u003cimg src=\"https://www.tensorflow.org/images/tf_logo_32px.png\" /\u003eView on TensorFlow.org\u003c/a\u003e\n",
-        "  \u003c/td\u003e\n",
-        "  \u003ctd\u003e\n",
-        "    \u003ca target=\"_blank\" href=\"https://colab.research.google.com/github/tensorflow/models/blob/master/official/colab/decoding_api_in_tf_nlp.ipynb\"\u003e\u003cimg src=\"https://www.tensorflow.org/images/colab_logo_32px.png\" /\u003eRun in Google Colab\u003c/a\u003e\n",
-        "  \u003c/td\u003e\n",
-        "  \u003ctd\u003e\n",
-        "    \u003ca target=\"_blank\" href=\"https://github.com/tensorflow/models/blob/master/official/colab/decoding_api_in_tf_nlp.ipynb\"\u003e\u003cimg src=\"https://www.tensorflow.org/images/GitHub-Mark-32px.png\" /\u003eView source on GitHub\u003c/a\u003e\n",
-        "  \u003c/td\u003e\n",
-        "  \u003ctd\u003e\n",
-        "    \u003ca href=\"https://storage.googleapis.com/tensorflow_docs/models/official/colab/decoding_api_in_tf_nlp.ipynb\"\u003e\u003cimg src=\"https://www.tensorflow.org/images/download_logo_32px.png\" /\u003eDownload notebook\u003c/a\u003e\n",
-        "  \u003c/td\u003e\n",
-        "\u003c/table\u003e"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "2j-xhrsVQOQT"
-      },
-      "outputs": [],
-      "source": [
-        "pip install  tf-models-nightly"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "BjP7zwxmskpY"
-      },
-      "outputs": [],
-      "source": [
-        "import os\n",
-        "\n",
-        "import numpy as np\n",
-        "import matplotlib.pyplot as plt\n",
-        "\n",
-        "import tensorflow as tf\n",
-        "\n",
-        "from official import nlp\n",
-        "from official.nlp.modeling.ops import sampling_module\n",
-        "from official.nlp.modeling.ops import beam_search"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "0AWgyo-IQ5sP"
-      },
-      "source": [
-        "# Decoding API\n",
-        "This API provides an interface to experiment with different decoding strategies used for auto-regressive models.\n",
-        "\n",
-        "1. The following sampling strategies are provided in sampling_module.py, which inherits from the base Decoding class:\n",
-        "  *   [top_p](https://arxiv.org/abs/1904.09751) : [github](https://github.com/tensorflow/models/blob/master/official/nlp/modeling/ops/sampling_module.py#L65) \n",
-        "\n",
-        "      This implementation chooses most probable logits with cumulative probabilities upto top_p.\n",
-        "\n",
-        "  *   [top_k](https://arxiv.org/pdf/1805.04833.pdf) : [github](https://github.com/tensorflow/models/blob/master/official/nlp/modeling/ops/sampling_module.py#L48)\n",
-        "\n",
-        "      At each timestep, this implementation samples from top-k logits based on their probability distribution\n",
-        "\n",
-        "  *   Greedy : [github](https://github.com/tensorflow/models/blob/master/official/nlp/modeling/ops/sampling_module.py#L26)\n",
-        "\n",
-        "      This implementation returns the top logits based on probabilities.\n",
-        "\n",
-        "2. Beam search is provided in beam_search.py. [github](https://github.com/tensorflow/models/blob/master/official/nlp/modeling/ops/beam_search.py)\n",
-        "\n",
-        "      This implementation reduces the risk of missing hidden high probability logits by keeping the most likely num_beams of logits at each time step and eventually choosing the logits that has the overall highest probability."
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "MfOj7oaBRQnS"
-      },
-      "source": [
-        "## Initialize Sampling Module in TF-NLP.\n",
-        "\n",
-        "\n",
-        "\u003e **symbols_to_logits_fn** : This is a closure implemented by the users of the API. The input to this closure will be  \n",
-        "```\n",
-        "Args:\n",
-        "  1] ids [batch_size, .. (index + 1 or 1 if padded_decode is True)],\n",
-        "  2] index [scalar] : current decoded step,\n",
-        "  3] cache [nested dictionary of tensors].\n",
-        "Returns:\n",
-        "  1] tensor for next-step logits [batch_size, vocab]\n",
-        "  2] the updated_cache [nested dictionary of tensors].\n",
-        "```\n",
-        "This closure calls the model to predict the logits for the 'index+1' step. The cache is used for faster decoding.\n",
-        "Here is a [reference](https://github.com/tensorflow/models/blob/master/official/nlp/modeling/ops/beam_search_test.py#L88) implementation for the above closure.\n",
-        "\n",
-        "\n",
-        "\u003e **length_normalization_fn** : Closure for returning length normalization parameter.\n",
-        "```\n",
-        "Args: \n",
-        "  1] length : scalar for decoded step index.\n",
-        "  2] dtype : data-type of output tensor\n",
-        "Returns:\n",
-        "  1] value of length normalization factor.\n",
-        "Example :\n",
-        "  def _length_norm(length, dtype):\n",
-        "    return tf.pow(((5. + tf.cast(length, dtype)) / 6.), 0.0)\n",
-        "```\n",
-        "\n",
-        "\u003e **vocab_size** : Output vocabulary size.\n",
-        "\n",
-        "\u003e **max_decode_length** : Scalar for total number of decoding steps.\n",
-        "\n",
-        "\u003e **eos_id** : Decoding will stop if all output decoded ids in the batch have this ID.\n",
-        "\n",
-        "\u003e **padded_decode** : Set this to True if running on TPU. Tensors are padded to max_decoding_length if this is True.\n",
-        "\n",
-        "\u003e **top_k** : top_k is enabled if this value is \u003e 1.\n",
-        "\n",
-        "\u003e **top_p** : top_p is enabled if this value is \u003e 0 and \u003c 1.0\n",
-        "\n",
-        "\u003e **sampling_temperature** : This is used to re-estimate the softmax output. Temperature skews the distribution towards high probability tokens and lowers the mass in tail distribution. Value has to be positive. Low temperature is equivalent to greedy and makes the distribution sharper, while high temperature makes it more flat.\n",
-        "\n",
-        "\u003e **enable_greedy** : By default, this is true and greedy decoding is enabled.\n"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "lV1RRp6ihnGX"
-      },
-      "source": [
-        "# Initialize the Model Hyper-parameters"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "eTsGp2gaKLdE"
-      },
-      "outputs": [],
-      "source": [
-        "params = {}\n",
-        "params['num_heads'] = 2\n",
-        "params['num_layers'] = 2\n",
-        "params['batch_size'] = 2\n",
-        "params['n_dims'] = 256\n",
-        "params['max_decode_length'] = 4"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "UGvmd0_dRFYI"
-      },
-      "source": [
-        "## What is a Cache?\n",
-        "In auto-regressive architectures like Transformer based [Encoder-Decoder](https://arxiv.org/abs/1706.03762) models, \n",
-        "Cache is used for fast sequential decoding.\n",
-        "It is a nested dictionary storing pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention blocks) for every layer.\n",
-        "\n",
-        "```\n",
-        "{\n",
-        "    'layer_%d' % layer: {\n",
-        "        'k': tf.zeros([params['batch_size'], params['max_decode_length'], params['num_heads'], params['n_dims']/params['num_heads']], dtype=tf.float32),\n",
-        "        'v': tf.zeros([params['batch_size'], params['max_decode_length'], params['num_heads'], params['n_dims']/params['num_heads']], dtype=tf.float32)\n",
-        "        } for layer in range(params['num_layers']),\n",
-        "    'model_specific_item' : Model specific tensor shape,\n",
-        "}\n",
-        "\n",
-        "```"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "CYXkoplAij01"
-      },
-      "source": [
-        "# Initialize cache. "
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "D6kfZOOKgkm1"
-      },
-      "outputs": [],
-      "source": [
-        "cache = {\n",
-        "    'layer_%d' % layer: {\n",
-        "        'k': tf.zeros([params['batch_size'], params['max_decode_length'], params['num_heads'], params['n_dims']/params['num_heads']], dtype=tf.float32),\n",
-        "        'v': tf.zeros([params['batch_size'], params['max_decode_length'], params['num_heads'], params['n_dims']/params['num_heads']], dtype=tf.float32)\n",
-        "        } for layer in range(params['num_layers'])\n",
-        "    }\n",
-        "print(\"cache key shape for layer 1 :\", cache['layer_1']['k'].shape)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "nNY3Xn8SiblP"
-      },
-      "source": [
-        "# Define closure for length normalization. **optional.**\n",
-        "\n",
-        "\n"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "T92ccAzlnGqh"
-      },
-      "outputs": [],
-      "source": [
-        "def length_norm(length, dtype):\n",
-        "  \"\"\"Return length normalization factor.\"\"\"\n",
-        "  return tf.pow(((5. + tf.cast(length, dtype)) / 6.), 0.0)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "syl7I5nURPgW"
-      },
-      "source": [
-        "# Create model_fn\n",
-        "  In practice, this will be replaced by an actual model implementation such as [here](https://github.com/tensorflow/models/blob/master/official/nlp/transformer/transformer.py#L236)\n",
-        "```\n",
-        "Args:\n",
-        "i : Step that is being decoded.\n",
-        "Returns:\n",
-        "  logit probabilities of size [batch_size, 1, vocab_size]\n",
-        "```\n",
-        "\n"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "AhzSkRisRdB6"
-      },
-      "outputs": [],
-      "source": [
-        "probabilities = tf.constant([[[0.3, 0.4, 0.3], [0.3, 0.3, 0.4],\n",
-        "                              [0.1, 0.1, 0.8], [0.1, 0.1, 0.8]],\n",
-        "                            [[0.2, 0.5, 0.3], [0.2, 0.7, 0.1],\n",
-        "                              [0.1, 0.1, 0.8], [0.1, 0.1, 0.8]]])\n",
-        "def model_fn(i):\n",
-        "  return probabilities[:, i, :]"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "DBMUkaVmVZBg"
-      },
-      "source": [
-        "# Initialize symbols_to_logits_fn\n"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "FAJ4CpbfVdjr"
-      },
-      "outputs": [],
-      "source": [
-        "def _symbols_to_logits_fn():\n",
-        "  \"\"\"Calculates logits of the next tokens.\"\"\"\n",
-        "  def symbols_to_logits_fn(ids, i, temp_cache):\n",
-        "    del ids\n",
-        "    logits = tf.cast(tf.math.log(model_fn(i)), tf.float32)\n",
-        "    return logits, temp_cache\n",
-        "  return symbols_to_logits_fn"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "R_tV3jyWVL47"
-      },
-      "source": [
-        "# Greedy \n",
-        "Greedy decoding selects the token id with the highest probability as its next id: $id_t = argmax_{w}P(id | id_{1:t-1})$ at each timestep $t$. The following sketch shows greedy decoding. "
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "aGt9idSkVQEJ"
-      },
-      "outputs": [],
-      "source": [
-        "greedy_obj = sampling_module.SamplingModule(\n",
-        "    length_normalization_fn=None,\n",
-        "    dtype=tf.float32,\n",
-        "    symbols_to_logits_fn=_symbols_to_logits_fn(),\n",
-        "    vocab_size=3,\n",
-        "    max_decode_length=params['max_decode_length'],\n",
-        "    eos_id=10,\n",
-        "    padded_decode=False)\n",
-        "ids, _ = greedy_obj.generate(\n",
-        "    initial_ids=tf.constant([9, 1]), initial_cache=cache)\n",
-        "print(\"Greedy Decoded Ids:\", ids)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "s4pTTsQXVz5O"
-      },
-      "source": [
-        "# top_k sampling\n",
-        "In *Top-K* sampling, the *K* most likely next token ids are filtered and the probability mass is redistributed among only those *K* ids. "
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "pCLWIn6GV5_G"
-      },
-      "outputs": [],
-      "source": [
-        "top_k_obj = sampling_module.SamplingModule(\n",
-        "    length_normalization_fn=length_norm,\n",
-        "    dtype=tf.float32,\n",
-        "    symbols_to_logits_fn=_symbols_to_logits_fn(),\n",
-        "    vocab_size=3,\n",
-        "    max_decode_length=params['max_decode_length'],\n",
-        "    eos_id=10,\n",
-        "    sample_temperature=tf.constant(1.0),\n",
-        "    top_k=tf.constant(3),\n",
-        "    padded_decode=False,\n",
-        "    enable_greedy=False)\n",
-        "ids, _ = top_k_obj.generate(\n",
-        "    initial_ids=tf.constant([9, 1]), initial_cache=cache)\n",
-        "print(\"top-k sampled Ids:\", ids)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "Jp3G-eE_WI4Y"
-      },
-      "source": [
-        "# top_p sampling\n",
-        "Instead of sampling only from the most likely *K* token ids, in *Top-p* sampling chooses from the smallest possible set of ids whose cumulative probability exceeds the probability *p*."
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "rEGdIWcuWILO"
-      },
-      "outputs": [],
-      "source": [
-        "top_p_obj = sampling_module.SamplingModule(\n",
-        "    length_normalization_fn=length_norm,\n",
-        "    dtype=tf.float32,\n",
-        "    symbols_to_logits_fn=_symbols_to_logits_fn(),\n",
-        "    vocab_size=3,\n",
-        "    max_decode_length=params['max_decode_length'],\n",
-        "    eos_id=10,\n",
-        "    sample_temperature=tf.constant(1.0),\n",
-        "    top_p=tf.constant(0.9),\n",
-        "    padded_decode=False,\n",
-        "    enable_greedy=False)\n",
-        "ids, _ = top_p_obj.generate(\n",
-        "    initial_ids=tf.constant([9, 1]), initial_cache=cache)\n",
-        "print(\"top-p sampled Ids:\", ids)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "2hcuyJ2VWjDz"
-      },
-      "source": [
-        "# Beam search decoding\n",
-        "Beam search reduces the risk of missing hidden high probability token ids by keeping the most likely num_beams of hypotheses at each time step and eventually choosing the hypothesis that has the overall highest probability. "
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "cJ3WzvSrWmSA"
-      },
-      "outputs": [],
-      "source": [
-        "beam_size = 2\n",
-        "params['batch_size'] = 1\n",
-        "beam_cache = {\n",
-        "    'layer_%d' % layer: {\n",
-        "        'k': tf.zeros([params['batch_size'], params['max_decode_length'], params['num_heads'], params['n_dims']], dtype=tf.float32),\n",
-        "        'v': tf.zeros([params['batch_size'], params['max_decode_length'], params['num_heads'], params['n_dims']], dtype=tf.float32)\n",
-        "        } for layer in range(params['num_layers'])\n",
-        "    }\n",
-        "print(\"cache key shape for layer 1 :\", beam_cache['layer_1']['k'].shape)\n",
-        "ids, _ = beam_search.sequence_beam_search(\n",
-        "    symbols_to_logits_fn=_symbols_to_logits_fn(),\n",
-        "    initial_ids=tf.constant([9], tf.int32),\n",
-        "    initial_cache=beam_cache,\n",
-        "    vocab_size=3,\n",
-        "    beam_size=beam_size,\n",
-        "    alpha=0.6,\n",
-        "    max_decode_length=params['max_decode_length'],\n",
-        "    eos_id=10,\n",
-        "    padded_decode=False,\n",
-        "    dtype=tf.float32)\n",
-        "print(\"Beam search ids:\", ids)"
-      ]
-    }
-  ],
-  "metadata": {
-    "accelerator": "GPU",
-    "colab": {
-      "collapsed_sections": [],
-      "name": "decoding_api_in_tf_nlp.ipynb",
-      "provenance": [],
-      "toc_visible": true
-    },
-    "kernelspec": {
-      "display_name": "Python 3",
-      "name": "python3"
-    }
-  },
-  "nbformat": 4,
-  "nbformat_minor": 0
-}
--- a/official/colab/nlp/nlp_modeling_library_intro.ipynb
+++ b/official/colab/nlp/nlp_modeling_library_intro.ipynb
-{
-  "cells": [
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "80xnUmoI7fBX"
-      },
-      "source": [
-        "##### Copyright 2020 The TensorFlow Authors."
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "cellView": "form",
-        "id": "8nvTnfs6Q692"
-      },
-      "outputs": [],
-      "source": [
-        "#@title Licensed under the Apache License, Version 2.0 (the \"License\");\n",
-        "# you may not use this file except in compliance with the License.\n",
-        "# You may obtain a copy of the License at\n",
-        "#\n",
-        "# https://www.apache.org/licenses/LICENSE-2.0\n",
-        "#\n",
-        "# Unless required by applicable law or agreed to in writing, software\n",
-        "# distributed under the License is distributed on an \"AS IS\" BASIS,\n",
-        "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n",
-        "# See the License for the specific language governing permissions and\n",
-        "# limitations under the License."
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "WmfcMK5P5C1G"
-      },
-      "source": [
-        "# Introduction to the TensorFlow Models NLP library"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "cH-oJ8R6AHMK"
-      },
-      "source": [
-        "\u003ctable class=\"tfo-notebook-buttons\" align=\"left\"\u003e\n",
-        "  \u003ctd\u003e\n",
-        "    \u003ca target=\"_blank\" href=\"https://www.tensorflow.org/official_models/nlp/nlp_modeling_library_intro\"\u003e\u003cimg src=\"https://www.tensorflow.org/images/tf_logo_32px.png\" /\u003eView on TensorFlow.org\u003c/a\u003e\n",
-        "  \u003c/td\u003e\n",
-        "  \u003ctd\u003e\n",
-        "    \u003ca target=\"_blank\" href=\"https://colab.research.google.com/github/tensorflow/models/blob/master/official/colab/nlp/nlp_modeling_library_intro.ipynb\"\u003e\u003cimg src=\"https://www.tensorflow.org/images/colab_logo_32px.png\" /\u003eRun in Google Colab\u003c/a\u003e\n",
-        "  \u003c/td\u003e\n",
-        "  \u003ctd\u003e\n",
-        "    \u003ca target=\"_blank\" href=\"https://github.com/tensorflow/models/blob/master/official/colab/nlp/nlp_modeling_library_intro.ipynb\"\u003e\u003cimg src=\"https://www.tensorflow.org/images/GitHub-Mark-32px.png\" /\u003eView source on GitHub\u003c/a\u003e\n",
-        "  \u003c/td\u003e\n",
-        "  \u003ctd\u003e\n",
-        "    \u003ca href=\"https://storage.googleapis.com/tensorflow_docs/models/official/colab/nlp/nlp_modeling_library_intro.ipynb\"\u003e\u003cimg src=\"https://www.tensorflow.org/images/download_logo_32px.png\" /\u003eDownload notebook\u003c/a\u003e\n",
-        "  \u003c/td\u003e\n",
-        "\u003c/table\u003e"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "0H_EFIhq4-MJ"
-      },
-      "source": [
-        "## Learning objectives\n",
-        "\n",
-        "In this Colab notebook, you will learn how to build transformer-based models for common NLP tasks including pretraining, span labelling and classification using the building blocks from [NLP modeling library](https://github.com/tensorflow/models/tree/master/official/nlp/modeling)."
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "2N97-dps_nUk"
-      },
-      "source": [
-        "## Install and import"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "459ygAVl_rg0"
-      },
-      "source": [
-        "### Install the TensorFlow Model Garden pip package\n",
-        "\n",
-        "*  `tf-models-official` is the stable Model Garden package. Note that it may not include the latest changes in the `tensorflow_models` github repo. To include latest changes, you may install `tf-models-nightly`,\n",
-        "which is the nightly Model Garden package created daily automatically.\n",
-        "*  `pip` will install all models and dependencies automatically."
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "Y-qGkdh6_sZc"
-      },
-      "outputs": [],
-      "source": [
-        "!pip install -q tf-models-official==2.4.0"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "e4huSSwyAG_5"
-      },
-      "source": [
-        "### Import Tensorflow and other libraries"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "jqYXqtjBAJd9"
-      },
-      "outputs": [],
-      "source": [
-        "import numpy as np\n",
-        "import tensorflow as tf\n",
-        "\n",
-        "from official.nlp import modeling\n",
-        "from official.nlp.modeling import layers, losses, models, networks"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "djBQWjvy-60Y"
-      },
-      "source": [
-        "## BERT pretraining model\n",
-        "\n",
-        "BERT ([Pre-training of Deep Bidirectional Transformers for Language Understanding](https://arxiv.org/abs/1810.04805)) introduced the method of pre-training language representations on a large text corpus and then using that model for downstream NLP tasks.\n",
-        "\n",
-        "In this section, we will learn how to build a model to pretrain BERT on the masked language modeling task and next sentence prediction task. For simplicity, we only show the minimum example and use dummy data."
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "MKuHVlsCHmiq"
-      },
-      "source": [
-        "### Build a `BertPretrainer` model wrapping `BertEncoder`\n",
-        "\n",
-        "The [BertEncoder](https://github.com/tensorflow/models/blob/master/official/nlp/modeling/networks/bert_encoder.py) implements the Transformer-based encoder as described in [BERT paper](https://arxiv.org/abs/1810.04805). It includes the embedding lookups and transformer layers, but not the masked language model or classification task networks.\n",
-        "\n",
-        "The [BertPretrainer](https://github.com/tensorflow/models/blob/master/official/nlp/modeling/models/bert_pretrainer.py) allows a user to pass in a transformer stack, and instantiates the masked language model and classification networks that are used to create the training objectives."
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "EXkcXz-9BwB3"
-      },
-      "outputs": [],
-      "source": [
-        "# Build a small transformer network.\n",
-        "vocab_size = 100\n",
-        "sequence_length = 16\n",
-        "network = modeling.networks.BertEncoder(\n",
-        "    vocab_size=vocab_size, num_layers=2, sequence_length=16)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "0NH5irV5KTMS"
-      },
-      "source": [
-        "Inspecting the encoder, we see it contains few embedding layers, stacked `Transformer` layers and are connected to three input layers:\n",
-        "\n",
-        "`input_word_ids`, `input_type_ids` and `input_mask`.\n"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "lZNoZkBrIoff"
-      },
-      "outputs": [],
-      "source": [
-        "tf.keras.utils.plot_model(network, show_shapes=True, dpi=48)"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "o7eFOZXiIl-b"
-      },
-      "outputs": [],
-      "source": [
-        "# Create a BERT pretrainer with the created network.\n",
-        "num_token_predictions = 8\n",
-        "bert_pretrainer = modeling.models.BertPretrainer(\n",
-        "    network, num_classes=2, num_token_predictions=num_token_predictions, output='predictions')"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "d5h5HT7gNHx_"
-      },
-      "source": [
-        "Inspecting the `bert_pretrainer`, we see it wraps the `encoder` with additional `MaskedLM` and `Classification` heads."
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "2tcNfm03IBF7"
-      },
-      "outputs": [],
-      "source": [
-        "tf.keras.utils.plot_model(bert_pretrainer, show_shapes=True, dpi=48)"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "F2oHrXGUIS0M"
-      },
-      "outputs": [],
-      "source": [
-        "# We can feed some dummy data to get masked language model and sentence output.\n",
-        "batch_size = 2\n",
-        "word_id_data = np.random.randint(vocab_size, size=(batch_size, sequence_length))\n",
-        "mask_data = np.random.randint(2, size=(batch_size, sequence_length))\n",
-        "type_id_data = np.random.randint(2, size=(batch_size, sequence_length))\n",
-        "masked_lm_positions_data = np.random.randint(2, size=(batch_size, num_token_predictions))\n",
-        "\n",
-        "outputs = bert_pretrainer(\n",
-        "    [word_id_data, mask_data, type_id_data, masked_lm_positions_data])\n",
-        "lm_output = outputs[\"masked_lm\"]\n",
-        "sentence_output = outputs[\"classification\"]\n",
-        "print(lm_output)\n",
-        "print(sentence_output)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "bnx3UCHniCS5"
-      },
-      "source": [
-        "### Compute loss\n",
-        "Next, we can use `lm_output` and `sentence_output` to compute `loss`."
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "k30H4Q86f52x"
-      },
-      "outputs": [],
-      "source": [
-        "masked_lm_ids_data = np.random.randint(vocab_size, size=(batch_size, num_token_predictions))\n",
-        "masked_lm_weights_data = np.random.randint(2, size=(batch_size, num_token_predictions))\n",
-        "next_sentence_labels_data = np.random.randint(2, size=(batch_size))\n",
-        "\n",
-        "mlm_loss = modeling.losses.weighted_sparse_categorical_crossentropy_loss(\n",
-        "    labels=masked_lm_ids_data,\n",
-        "    predictions=lm_output,\n",
-        "    weights=masked_lm_weights_data)\n",
-        "sentence_loss = modeling.losses.weighted_sparse_categorical_crossentropy_loss(\n",
-        "    labels=next_sentence_labels_data,\n",
-        "    predictions=sentence_output)\n",
-        "loss = mlm_loss + sentence_loss\n",
-        "print(loss)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "wrmSs8GjHxVw"
-      },
-      "source": [
-        "With the loss, you can optimize the model.\n",
-        "After training, we can save the weights of TransformerEncoder for the downstream fine-tuning tasks. Please see [run_pretraining.py](https://github.com/tensorflow/models/blob/master/official/nlp/bert/run_pretraining.py) for the full example.\n",
-        "\n"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "k8cQVFvBCV4s"
-      },
-      "source": [
-        "## Span labeling model\n",
-        "\n",
-        "Span labeling is the task to assign labels to a span of the text, for example, label a span of text as the answer of a given question.\n",
-        "\n",
-        "In this section, we will learn how to build a span labeling model. Again, we use dummy data for simplicity."
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "xrLLEWpfknUW"
-      },
-      "source": [
-        "### Build a BertSpanLabeler wrapping BertEncoder\n",
-        "\n",
-        "[BertSpanLabeler](https://github.com/tensorflow/models/blob/master/official/nlp/modeling/models/bert_span_labeler.py) implements a simple single-span start-end predictor (that is, a model that predicts two values: a start token index and an end token index), suitable for SQuAD-style tasks.\n",
-        "\n",
-        "Note that `BertSpanLabeler` wraps a `BertEncoder`, the weights of which can be restored from the above pretraining model.\n"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "B941M4iUCejO"
-      },
-      "outputs": [],
-      "source": [
-        "network = modeling.networks.BertEncoder(\n",
-        "        vocab_size=vocab_size, num_layers=2, sequence_length=sequence_length)\n",
-        "\n",
-        "# Create a BERT trainer with the created network.\n",
-        "bert_span_labeler = modeling.models.BertSpanLabeler(network)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "QpB9pgj4PpMg"
-      },
-      "source": [
-        "Inspecting the `bert_span_labeler`, we see it wraps the encoder with additional `SpanLabeling` that outputs `start_position` and `end_postion`."
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "RbqRNJCLJu4H"
-      },
-      "outputs": [],
-      "source": [
-        "tf.keras.utils.plot_model(bert_span_labeler, show_shapes=True, dpi=48)"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "fUf1vRxZJwio"
-      },
-      "outputs": [],
-      "source": [
-        "# Create a set of 2-dimensional data tensors to feed into the model.\n",
-        "word_id_data = np.random.randint(vocab_size, size=(batch_size, sequence_length))\n",
-        "mask_data = np.random.randint(2, size=(batch_size, sequence_length))\n",
-        "type_id_data = np.random.randint(2, size=(batch_size, sequence_length))\n",
-        "\n",
-        "# Feed the data to the model.\n",
-        "start_logits, end_logits = bert_span_labeler([word_id_data, mask_data, type_id_data])\n",
-        "print(start_logits)\n",
-        "print(end_logits)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "WqhgQaN1lt-G"
-      },
-      "source": [
-        "### Compute loss\n",
-        "With `start_logits` and `end_logits`, we can compute loss:"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "waqs6azNl3Nn"
-      },
-      "outputs": [],
-      "source": [
-        "start_positions = np.random.randint(sequence_length, size=(batch_size))\n",
-        "end_positions = np.random.randint(sequence_length, size=(batch_size))\n",
-        "\n",
-        "start_loss = tf.keras.losses.sparse_categorical_crossentropy(\n",
-        "    start_positions, start_logits, from_logits=True)\n",
-        "end_loss = tf.keras.losses.sparse_categorical_crossentropy(\n",
-        "    end_positions, end_logits, from_logits=True)\n",
-        "\n",
-        "total_loss = (tf.reduce_mean(start_loss) + tf.reduce_mean(end_loss)) / 2\n",
-        "print(total_loss)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "Zdf03YtZmd_d"
-      },
-      "source": [
-        "With the `loss`, you can optimize the model. Please see [run_squad.py](https://github.com/tensorflow/models/blob/master/official/nlp/bert/run_squad.py) for the full example."
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "0A1XnGSTChg9"
-      },
-      "source": [
-        "## Classification model\n",
-        "\n",
-        "In the last section, we show how to build a text classification model.\n"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "MSK8OpZgnQa9"
-      },
-      "source": [
-        "### Build a BertClassifier model wrapping BertEncoder\n",
-        "\n",
-        "[BertClassifier](https://github.com/tensorflow/models/blob/master/official/nlp/modeling/models/bert_classifier.py) implements a [CLS] token classification model containing a single classification head."
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "cXXCsffkCphk"
-      },
-      "outputs": [],
-      "source": [
-        "network = modeling.networks.BertEncoder(\n",
-        "        vocab_size=vocab_size, num_layers=2, sequence_length=sequence_length)\n",
-        "\n",
-        "# Create a BERT trainer with the created network.\n",
-        "num_classes = 2\n",
-        "bert_classifier = modeling.models.BertClassifier(\n",
-        "    network, num_classes=num_classes)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "8tZKueKYP4bB"
-      },
-      "source": [
-        "Inspecting the `bert_classifier`, we see it wraps the `encoder` with additional `Classification` head."
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "snlutm9ZJgEZ"
-      },
-      "outputs": [],
-      "source": [
-        "tf.keras.utils.plot_model(bert_classifier, show_shapes=True, dpi=48)"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "yyHPHsqBJkCz"
-      },
-      "outputs": [],
-      "source": [
-        "# Create a set of 2-dimensional data tensors to feed into the model.\n",
-        "word_id_data = np.random.randint(vocab_size, size=(batch_size, sequence_length))\n",
-        "mask_data = np.random.randint(2, size=(batch_size, sequence_length))\n",
-        "type_id_data = np.random.randint(2, size=(batch_size, sequence_length))\n",
-        "\n",
-        "# Feed the data to the model.\n",
-        "logits = bert_classifier([word_id_data, mask_data, type_id_data])\n",
-        "print(logits)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "w--a2mg4nzKm"
-      },
-      "source": [
-        "### Compute loss\n",
-        "\n",
-        "With `logits`, we can compute `loss`:"
-      ]
-    },
-    {
-      "cell_type": "code",
-      "execution_count": null,
-      "metadata": {
-        "id": "9X0S1DoFn_5Q"
-      },
-      "outputs": [],
-      "source": [
-        "labels = np.random.randint(num_classes, size=(batch_size))\n",
-        "\n",
-        "loss = tf.keras.losses.sparse_categorical_crossentropy(\n",
-        "    labels, logits, from_logits=True)\n",
-        "print(loss)"
-      ]
-    },
-    {
-      "cell_type": "markdown",
-      "metadata": {
-        "id": "mzBqOylZo3og"
-      },
-      "source": [
-        "With the `loss`, you can optimize the model. Please see [run_classifier.py](https://github.com/tensorflow/models/blob/master/official/nlp/bert/run_classifier.py) or the colab [fine_tuning_bert.ipynb](https://github.com/tensorflow/models/blob/master/official/colab/fine_tuning_bert.ipynb) for the full example."
-      ]
-    }
-  ],
-  "metadata": {
-    "colab": {
-      "collapsed_sections": [],
-      "name": "Introduction to the TensorFlow Models NLP library",
-      "private_outputs": true,
-      "provenance": [],
-      "toc_visible": true
-    },
-    "kernelspec": {
-      "display_name": "Python 3",
-      "name": "python3"
-    }
-  },
-  "nbformat": 4,
-  "nbformat_minor": 0
-}
--- a/official/common/__init__.py
+++ b/official/common/__init__.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.

--- a/official/common/dataset_fn.py
+++ b/official/common/dataset_fn.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.
@@ -28,7 +28,8 @@
 # ==============================================================================
 """Utility library for picking an appropriate dataset function."""

-from typing import Any, Callable, Union, Type
+import functools
+from typing import Any, Callable, Type, Union

 import tensorflow as tf

@@ -38,5 +39,6 @@ PossibleDatasetType = Union[Type[tf.data.Dataset], Callable[[tf.Tensor], Any]]
 def pick_dataset_fn(file_type: str) -> PossibleDatasetType:
  if file_type == 'tfrecord':
    return tf.data.TFRecordDataset
-
+  if file_type == 'tfrecord_compressed':
+    return functools.partial(tf.data.TFRecordDataset, compression_type='GZIP')
  raise ValueError('Unrecognized file_type: {}'.format(file_type))
--- a/official/common/distribute_utils.py
+++ b/official/common/distribute_utils.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.
@@ -96,7 +96,7 @@ def get_distribution_strategy(distribution_strategy="mirrored",
                              num_packs=1,
                              tpu_address=None,
                              **kwargs):
-  """Return a DistributionStrategy for running the model.
+  """Return a Strategy for running the model.

  Args:
    distribution_strategy: a string specifying which distribution strategy to
@@ -119,7 +119,7 @@ def get_distribution_strategy(distribution_strategy="mirrored",
    **kwargs: Additional kwargs for internal usages.

  Returns:
-    tf.distribute.DistibutionStrategy object.
+    tf.distribute.Strategy object.
  Raises:
    ValueError: if `distribution_strategy` is "off" or "one_device" and
      `num_gpus` is larger than 1; or `num_gpus` is negative or if

--- a/official/common/distribute_utils_test.py
+++ b/official/common/distribute_utils_test.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.

--- a/official/common/flags.py
+++ b/official/common/flags.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.
@@ -45,7 +45,8 @@ def define_flags():
      default=None,
      enum_values=[
          'train', 'eval', 'train_and_eval', 'continuous_eval',
-          'continuous_train_and_eval', 'train_and_validate'
+          'continuous_train_and_eval', 'train_and_validate',
+          'train_and_post_eval'
      ],
      help='Mode to run: `train`, `eval`, `train_and_eval`, '
      '`continuous_eval`, `continuous_train_and_eval` and '

--- a/official/common/registry_imports.py
+++ b/official/common/registry_imports.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.
@@ -14,7 +14,7 @@

 """All necessary imports for registration."""
 # pylint: disable=unused-import
+from official import vision
 from official.nlp import tasks
 from official.nlp.configs import experiment_configs
 from official.utils.testing import mock_task
-from official.vision import beta
--- a/official/common/streamz_counters.py
+++ b/official/common/streamz_counters.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.

--- a/official/core/__init__.py
+++ b/official/core/__init__.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.
@@ -12,3 +12,20 @@
 # See the License for the specific language governing permissions and
 # limitations under the License.

+"""Core is shared by both `nlp` and `vision`."""
+
+from official.core import actions
+from official.core import base_task
+from official.core import base_trainer
+from official.core import config_definitions
+from official.core import exp_factory
+from official.core import export_base
+from official.core import file_writers
+from official.core import input_reader
+from official.core import registry
+from official.core import savedmodel_checkpoint_manager
+from official.core import task_factory
+from official.core import tf_example_builder
+from official.core import tf_example_feature_key
+from official.core import train_lib
+from official.core import train_utils
--- a/official/core/actions.py
+++ b/official/core/actions.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.
@@ -21,7 +21,6 @@ from absl import logging
 import gin
 import orbit
 import tensorflow as tf
-import tensorflow_model_optimization as tfmot

 from official.core import base_trainer
 from official.core import config_definitions
@@ -52,6 +51,8 @@ class PruningAction:
      optimizer: `tf.keras.optimizers.Optimizer` optimizer instance used for
        training. This will be used to find the current training steps.
    """
+    # TODO(b/221490190): Avoid local import when the bug is fixed.
+    import tensorflow_model_optimization as tfmot  # pylint: disable=g-import-not-at-top
    self._optimizer = optimizer
    self.update_pruning_step = tfmot.sparsity.keras.UpdatePruningStep()
    self.update_pruning_step.set_model(model)
@@ -201,7 +202,7 @@ def get_train_actions(
  """Gets train actions for TFM trainer."""
  train_actions = []
  # Adds pruning callback actions.
-  if hasattr(params.task, 'pruning'):
+  if hasattr(params.task, 'pruning') and params.task.pruning:
    train_actions.append(
        PruningAction(
            export_dir=model_dir,

--- a/official/core/actions_test.py
+++ b/official/core/actions_test.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.

--- a/official/core/base_task.py
+++ b/official/core/base_task.py
-# Copyright 2021 The TensorFlow Authors. All Rights Reserved.
+# Copyright 2022 The TensorFlow Authors. All Rights Reserved.
 #
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.
@@ -14,6 +14,7 @@

 """Defines the base task abstraction."""
 import abc
+import functools
 from typing import Optional

 from absl import logging
@@ -22,9 +23,12 @@ import tensorflow as tf
 from official.core import config_definitions
 from official.modeling import optimization
 from official.modeling import performance
+from official.modeling.privacy import configs
+from official.modeling.privacy import ops

 OptimizationConfig = optimization.OptimizationConfig
 RuntimeConfig = config_definitions.RuntimeConfig
+DifferentialPrivacyConfig = configs.DifferentialPrivacyConfig


 class Task(tf.Module, metaclass=abc.ABCMeta):
@@ -65,18 +69,35 @@ class Task(tf.Module, metaclass=abc.ABCMeta):

  @classmethod
  def create_optimizer(cls, optimizer_config: OptimizationConfig,
-                       runtime_config: Optional[RuntimeConfig] = None):
+                       runtime_config: Optional[RuntimeConfig] = None,
+                       dp_config: Optional[DifferentialPrivacyConfig] = None):
    """Creates an TF optimizer from configurations.

    Args:
      optimizer_config: the parameters of the Optimization settings.
      runtime_config: the parameters of the runtime.
+      dp_config: the parameter of differential privacy.

    Returns:
      A tf.optimizers.Optimizer object.
    """
+    gradient_transformers = None
+    if dp_config is not None:
+      logging.info("Adding differential privacy transform with config %s.",
+                   dp_config.as_dict())
+      noise_stddev = dp_config.clipping_norm * dp_config.noise_multiplier
+      gradient_transformers = [
+          functools.partial(
+              ops.clip_l2_norm, l2_norm_clip=dp_config.clipping_norm),
+          functools.partial(
+              ops.add_noise, noise_stddev=noise_stddev)
+      ]
+
    opt_factory = optimization.OptimizerFactory(optimizer_config)
-    optimizer = opt_factory.build_optimizer(opt_factory.build_learning_rate())
+    optimizer = opt_factory.build_optimizer(
+        opt_factory.build_learning_rate(),
+        gradient_transformers=gradient_transformers
+        )
    # Configuring optimizer when loss_scale is set in runtime config. This helps
    # avoiding overflow/underflow for float16 computations.
    if runtime_config:
@@ -101,9 +122,11 @@ class Task(tf.Module, metaclass=abc.ABCMeta):
    ckpt_dir_or_file = self.task_config.init_checkpoint
    logging.info("Trying to load pretrained checkpoint from %s",
                 ckpt_dir_or_file)
-    if tf.io.gfile.isdir(ckpt_dir_or_file):
+    if ckpt_dir_or_file and tf.io.gfile.isdir(ckpt_dir_or_file):
      ckpt_dir_or_file = tf.train.latest_checkpoint(ckpt_dir_or_file)
    if not ckpt_dir_or_file:
+      logging.info("No checkpoint file found from %s. Will not load.",
+                   ckpt_dir_or_file)
      return

    if hasattr(model, "checkpoint_items"):