Spaces:

xu-song
/

tokenizer-arena

Running

App Files Files Community

xu-song commited on Sep 5, 2023

Commit

7156337

•

1 Parent(s): 309a593

update

Browse files

Files changed (9) hide show

app.py +5 -4
vocab/gpt_4/__init__.py +42 -0
vocab/gpt_neox_chinese_v1/20B_tokenizer_chinese.mock.json +0 -0
vocab/gpt_neox_chinese_v1/README.md +52 -0
vocab/gpt_neox_chinese_v1/mock.py +17 -0
vocab/gpt_neox_chinese_v1/tokenizer/__init__.py +16 -0
vocab/gpt_neox_chinese_v1/tokenizer/gpt2_tokenization.py +368 -0
vocab/gpt_neox_chinese_v1/tokenizer/tokenizer.py +402 -0
vocab/gpt_neox_chinese_v1/tokenizer/train_tokenizer.py +126 -0

app.py CHANGED Viewed

@@ -15,7 +15,8 @@
 - 英文 utf-8编码
 - 词典支持下载
 - 中文字词统计，是否要包括 _ G 等字符
 plots
@@ -137,7 +138,7 @@ with gr.Blocks(css="style.css") as demo:
                         # )
         # https://www.onlinewebfonts.com/icon/418591
         gr.Image("images/VS.svg", scale=1, show_label=False,
-                 show_download_button=False, container=False,
                  show_share_button=False)
         with gr.Column(scale=6):
             with gr.Group():
@@ -154,9 +155,9 @@ with gr.Blocks(css="style.css") as demo:
                             lines=1,
                             elem_classes="statistics"
                         )
-                        stats_zh_token_size_2 = gr.TextArea(  # 中文单子数，
                             value=default_stats_zh_token_size_2,
-                            label="ZH char/word",
                             lines=1,
                             elem_classes="statistics"
                         )

 - 英文 utf-8编码
 - 词典支持下载
 - 中文字词统计，是否要包括 _ G 等字符
+- baichuan的单字数量怎么两万多个？
+- gpt4
 plots
                         # )
         # https://www.onlinewebfonts.com/icon/418591
         gr.Image("images/VS.svg", scale=1, show_label=False,
+                 show_download_button=True, container=False,
                  show_share_button=False)
         with gr.Column(scale=6):
             with gr.Group():
                             lines=1,
                             elem_classes="statistics"
                         )
+                        stats_zh_token_size_2 = gr.TextArea(
                             value=default_stats_zh_token_size_2,
+                            label="ZH char/word",  # 中文字/词
                             lines=1,
                             elem_classes="statistics"
                         )

vocab/gpt_4/__init__.py ADDED Viewed

	@@ -0,0 +1,42 @@

+import tiktoken
+from tiktoken import Encoding
+tokenizer = tiktoken.encoding_for_model('gpt-4')
+tokenizer.vocab_size = tokenizer.n_vocab
+def decode(self, tokens, errors="replace"):
+# def decode(self, tokens: list[int], errors: str = "replace") -> str:
+    try:
+        decode_str = self._core_bpe.decode_bytes(tokens).decode("utf-8", errors=errors)
+    except:
+        decode_str = "null"
+    return decode_str
+def convert_ids_to_tokens(self, tokens):
+    return tokenizer.decode_tokens_bytes(tokens)
+def get_vocab(self):
+    """Returns vocab as a dict"""
+    vocab = {}
+    for i in range(self.vocab_size):
+        try:
+            token_byte = self.convert_ids_to_tokens([i])[0]
+            token_str = token_byte.decode("utf-8")
+            vocab[token_str] = i
+        except KeyError:
+            print("gpt_35_turbo decode KeyError", i)
+        except UnicodeDecodeError:
+            print("gpt_35_turbo decode UnicodeDecodeError", i, str(token_byte))
+    # vocab.update(self.added_tokens_encoder)
+    return vocab
+Encoding.decode = decode
+Encoding.convert_ids_to_tokens = convert_ids_to_tokens
+Encoding.get_vocab = get_vocab

vocab/gpt_neox_chinese_v1/20B_tokenizer_chinese.mock.json ADDED Viewed

The diff for this file is too large to render. See raw diff

vocab/gpt_neox_chinese_v1/README.md CHANGED Viewed

	@@ -10,3 +10,55 @@ Vocab size: 54634
10
11	https://github.com/huggingface/transformers/blob/main/src/transformers/models/gpt_neox_japanese/tokenization_gpt_neox_japanese.py
12

 https://github.com/huggingface/transformers/blob/main/src/transformers/models/gpt_neox_japanese/tokenization_gpt_neox_japanese.py
+## 20B
+[configs/20B.yml](https://github.com/EleutherAI/gpt-neox/blob/main/configs/20B.yml#L7)
+```
+  "vocab-file": "./20B_checkpoints/20B_tokenizer.json",
+```
+Vocab size: 50277
+self.padded_vocab_size = 50304
+padded vocab (size: 50277) with 27 dummy tokens (new size: 50304)
+## 词典
+见 convert_vocab_to_txt.py
+```
+{"id": 13609, "token": "\u00e4\u00b8\u0143", "token_decode": "\u4e2d"}	中
+# 多个符号拼接在一起的
+{"id": 13663, "token": ".*]{}", "token_decode": ".*]{}"}	.*]{}
+# ss
+```
+## 中文支持
+基本没有OOV。
+gpt-neox是在800G英文数据集上训练的，为啥词典支持中文？因为是byte-level BPE
+```
+丁 [3218, 212]
+七 [3218, 214]
+万 [3218, 218]
+诀 [11894, 211]
+证 [11894, 212]
+```
+编码长度统计： Counter({2: 4190, 3: 1295, 1: 285})
+平均编码长度： 2.1750433275563257
+## ss

vocab/gpt_neox_chinese_v1/mock.py ADDED Viewed

	@@ -0,0 +1,17 @@

+import copy
+import json
+input_path = "20B_tokenizer_chinese.json"
+tokenizer = json.load(open(input_path, "r", encoding="utf-8"))
+vocab = tokenizer["model"]["vocab"]
+for k, v in copy.deepcopy(vocab).items():
+    vocab[str(v)] = v
+    vocab.pop(k)
+out_path = input_path.replace(".json", ".mock.json")
+with open(out_path, "w", encoding="utf-8") as f_out:
+    f_out.write(json.dumps(tokenizer, ensure_ascii=False, indent=2))

vocab/gpt_neox_chinese_v1/tokenizer/__init__.py ADDED Viewed

	@@ -0,0 +1,16 @@

+# Copyright (c) 2020, NVIDIA CORPORATION.  All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from .tokenizer import build_tokenizer

vocab/gpt_neox_chinese_v1/tokenizer/gpt2_tokenization.py ADDED Viewed

	@@ -0,0 +1,368 @@

+# Copyright (c) 2021, EleutherAI
+# This file is based on code by the authors denoted below and has been modified from its original version.
+#
+# Copyright 2018 The Open AI Team Authors and The HuggingFace Inc. team.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Tokenization classes for OpenAI GPT."""
+from __future__ import absolute_import, division, print_function, unicode_literals
+import sys
+import json
+import logging
+import os
+import regex as re
+from io import open
+from functools import lru_cache
+logger = logging.getLogger(__name__)
+PRETRAINED_VOCAB_ARCHIVE_MAP = {
+    "gpt2": "https://s3.amazonaws.com/models.huggingface.co/bert/gpt2-vocab.json",
+}
+PRETRAINED_MERGES_ARCHIVE_MAP = {
+    "gpt2": "https://s3.amazonaws.com/models.huggingface.co/bert/gpt2-merges.txt",
+}
+PRETRAINED_VOCAB_POSITIONAL_EMBEDDINGS_SIZE_MAP = {
+    "gpt2": 1024,
+}
+VOCAB_NAME = "vocab.json"
+MERGES_NAME = "merges.txt"
+SPECIAL_TOKENS_NAME = "special_tokens.txt"
+@lru_cache()
+def bytes_to_unicode():
+    """
+    Returns list of utf-8 byte and a corresponding list of unicode strings.
+    The reversible bpe codes work on unicode strings.
+    This means you need a large # of unicode characters in your vocab if you want to avoid UNKs.
+    When you're at something like a 10B token dataset you end up needing around 5K for decent coverage.
+    This is a significant percentage of your normal, say, 32K bpe vocab.
+    To avoid that, we want lookup tables between utf-8 bytes and unicode strings.
+    And avoids mapping to whitespace/control characters the bpe code barfs on.
+    """
+    _chr = unichr if sys.version_info[0] == 2 else chr
+    bs = (
+        list(range(ord("!"), ord("~") + 1))
+        + list(range(ord("¡"), ord("¬") + 1))
+        + list(range(ord("®"), ord("ÿ") + 1))
+    )
+    cs = bs[:]
+    n = 0
+    for b in range(2**8):
+        if b not in bs:
+            bs.append(b)
+            cs.append(2**8 + n)
+            n += 1
+    cs = [_chr(n) for n in cs]
+    return dict(zip(bs, cs))
+def get_pairs(word):
+    """Return set of symbol pairs in a word.
+    Word is represented as tuple of symbols (symbols being variable-length strings).
+    """
+    pairs = set()
+    prev_char = word[0]
+    for char in word[1:]:
+        pairs.add((prev_char, char))
+        prev_char = char
+    return pairs
+class GPT2Tokenizer(object):
+    """
+    GPT-2 BPE tokenizer. Peculiarities:
+        - Byte-level BPE
+    """
+    @classmethod
+    def from_pretrained(
+        cls, pretrained_model_name_or_path, cache_dir=None, *inputs, **kwargs
+    ):
+        """
+        Instantiate a PreTrainedBertModel from a pre-trained model file.
+        Download and cache the pre-trained model file if needed.
+        """
+        if pretrained_model_name_or_path in PRETRAINED_VOCAB_ARCHIVE_MAP:
+            vocab_file = PRETRAINED_VOCAB_ARCHIVE_MAP[pretrained_model_name_or_path]
+            merges_file = PRETRAINED_MERGES_ARCHIVE_MAP[pretrained_model_name_or_path]
+            special_tokens_file = None
+        else:
+            vocab_file = os.path.join(pretrained_model_name_or_path, VOCAB_NAME)
+            merges_file = os.path.join(pretrained_model_name_or_path, MERGES_NAME)
+            special_tokens_file = os.path.join(
+                pretrained_model_name_or_path, SPECIAL_TOKENS_NAME
+            )
+            if not os.path.exists(special_tokens_file):
+                special_tokens_file = None
+            else:
+                logger.info(
+                    "loading special tokens file {}".format(special_tokens_file)
+                )
+        # redirect to the cache, if necessary
+        try:
+            from .file_utils import cached_path
+            resolved_vocab_file = cached_path(vocab_file, cache_dir=cache_dir)
+            resolved_merges_file = cached_path(merges_file, cache_dir=cache_dir)
+        except EnvironmentError:
+            logger.error(
+                "Model name '{}' was not found in model name list ({}). "
+                "We assumed '{}' was a path or url but couldn't find files {} and {} "
+                "at this path or url.".format(
+                    pretrained_model_name_or_path,
+                    ", ".join(PRETRAINED_VOCAB_ARCHIVE_MAP.keys()),
+                    pretrained_model_name_or_path,
+                    vocab_file,
+                    merges_file,
+                )
+            )
+            return None
+        if resolved_vocab_file == vocab_file and resolved_merges_file == merges_file:
+            logger.info("loading vocabulary file {}".format(vocab_file))
+            logger.info("loading merges file {}".format(merges_file))
+        else:
+            logger.info(
+                "loading vocabulary file {} from cache at {}".format(
+                    vocab_file, resolved_vocab_file
+                )
+            )
+            logger.info(
+                "loading merges file {} from cache at {}".format(
+                    merges_file, resolved_merges_file
+                )
+            )
+        if (
+            pretrained_model_name_or_path
+            in PRETRAINED_VOCAB_POSITIONAL_EMBEDDINGS_SIZE_MAP
+        ):
+            # if we're using a pretrained model, ensure the tokenizer won't index sequences longer
+            # than the number of positional embeddings
+            max_len = PRETRAINED_VOCAB_POSITIONAL_EMBEDDINGS_SIZE_MAP[
+                pretrained_model_name_or_path
+            ]
+            kwargs["max_len"] = min(kwargs.get("max_len", int(1e12)), max_len)
+        # Instantiate tokenizer.
+        if special_tokens_file and "special_tokens" not in kwargs:
+            special_tokens = (
+                open(special_tokens_file, encoding="utf-8").read().split("\n")[:-1]
+            )
+        else:
+            special_tokens = kwargs.pop("special_tokens", [])
+        tokenizer = cls(
+            resolved_vocab_file,
+            resolved_merges_file,
+            special_tokens=special_tokens,
+            *inputs,
+            **kwargs
+        )
+        return tokenizer
+    def __init__(
+        self,
+        vocab_file,
+        merges_file,
+        errors="replace",
+        special_tokens=None,
+        max_len=None,
+    ):
+        self.max_len = max_len if max_len is not None else int(1e12)
+        self.encoder = json.load(open(vocab_file))
+        self.decoder = {v: k for k, v in self.encoder.items()}
+        self.errors = errors  # how to handle errors in decoding
+        self.byte_encoder = bytes_to_unicode()
+        self.byte_decoder = {v: k for k, v in self.byte_encoder.items()}
+        bpe_data = open(merges_file, encoding="utf-8").read().split("\n")[1:-1]
+        bpe_merges = [tuple(merge.split()) for merge in bpe_data]
+        self.bpe_ranks = dict(zip(bpe_merges, range(len(bpe_merges))))
+        # Should haved added re.IGNORECASE so BPE merges can happen for
+        # capitalized versions of contractions
+        self.pat = re.compile(
+            r"""'s|'t|'re|'ve|'m|'ll|'d| ?\p{L}+| ?\p{N}+| ?[^\s\p{L}\p{N}]+|\s+(?!\S)|\s+"""
+        )
+        self.special_tokens = {}
+        self.special_tokens_decoder = {}
+        self.set_special_tokens(special_tokens)
+    def __len__(self):
+        return len(self.encoder) + len(self.special_tokens)
+    def set_special_tokens(self, special_tokens):
+        """Add a list of additional tokens to the encoder.
+        The additional tokens are indexed starting from the last index of the
+        current vocabulary in the order of the `special_tokens` list.
+        """
+        if not special_tokens:
+            self.special_tokens = {}
+            self.special_tokens_decoder = {}
+            return
+        self.special_tokens = dict(
+            (tok, len(self.encoder) + i) for i, tok in enumerate(special_tokens)
+        )
+        self.special_tokens_decoder = {v: k for k, v in self.special_tokens.items()}
+        logger.info("Special tokens {}".format(self.special_tokens))
+    @lru_cache(maxsize=131072)
+    def bpe(self, token):
+        word = tuple(token)
+        pairs = get_pairs(word)
+        if not pairs:
+            return token
+        while True:
+            bigram = min(pairs, key=lambda pair: self.bpe_ranks.get(pair, float("inf")))
+            if bigram not in self.bpe_ranks:
+                break
+            first, second = bigram
+            new_word = []
+            i = 0
+            while i < len(word):
+                try:
+                    j = word.index(first, i)
+                    new_word.extend(word[i:j])
+                    i = j
+                except BaseException:
+                    new_word.extend(word[i:])
+                    break
+                if word[i] == first and i < len(word) - 1 and word[i + 1] == second:
+                    new_word.append(first + second)
+                    i += 2
+                else:
+                    new_word.append(word[i])
+                    i += 1
+            new_word = tuple(new_word)
+            word = new_word
+            if len(word) == 1:
+                break
+            else:
+                pairs = get_pairs(word)
+        word = " ".join(word)
+        return word
+    def tokenize(self, text):
+        """Tokenize a string."""
+        bpe_tokens = []
+        for token in re.findall(self.pat, text):
+            if sys.version_info[0] == 2:
+                token = "".join(self.byte_encoder[ord(b)] for b in token)
+            else:
+                token = "".join(self.byte_encoder[b] for b in token.encode("utf-8"))
+            bpe_tokens.extend(bpe_token for bpe_token in self.bpe(token).split(" "))
+        return bpe_tokens
+    def convert_tokens_to_ids(self, tokens):
+        """Converts a sequence of tokens into ids using the vocab."""
+        ids = []
+        if isinstance(tokens, str) or (
+            sys.version_info[0] == 2 and isinstance(tokens, unicode)
+        ):
+            if tokens in self.special_tokens:
+                return self.special_tokens[tokens]
+            else:
+                return self.encoder.get(tokens, 0)
+        for token in tokens:
+            if token in self.special_tokens:
+                ids.append(self.special_tokens[token])
+            else:
+                ids.append(self.encoder.get(token, 0))
+        if len(ids) > self.max_len:
+            logger.warning(
+                "Token indices sequence length is longer than the specified maximum "
+                " sequence length for this OpenAI GPT model ({} > {}). Running this"
+                " sequence through the model will result in indexing errors".format(
+                    len(ids), self.max_len
+                )
+            )
+        return ids
+    def convert_ids_to_tokens(self, ids, skip_special_tokens=False):
+        """Converts a sequence of ids in BPE tokens using the vocab."""
+        tokens = []
+        for i in ids:
+            if i in self.special_tokens_decoder:
+                if not skip_special_tokens:
+                    tokens.append(self.special_tokens_decoder[i])
+            else:
+                tokens.append(self.decoder[i])
+        return tokens
+    def encode(self, text):
+        return self.convert_tokens_to_ids(self.tokenize(text))
+    def decode(self, tokens):
+        text = "".join([self.decoder[token] for token in tokens])
+        text = bytearray([self.byte_decoder[c] for c in text]).decode(
+            "utf-8", errors=self.errors
+        )
+        return text
+    def save_vocabulary(self, vocab_path):
+        """Save the tokenizer vocabulary and merge files to a directory."""
+        if not os.path.isdir(vocab_path):
+            logger.error(
+                "Vocabulary path ({}) should be a directory".format(vocab_path)
+            )
+            return
+        vocab_file = os.path.join(vocab_path, VOCAB_NAME)
+        merge_file = os.path.join(vocab_path, MERGES_NAME)
+        special_tokens_file = os.path.join(vocab_path, SPECIAL_TOKENS_NAME)
+        with open(vocab_file, "w", encoding="utf-8") as f:
+            f.write(json.dumps(self.encoder, ensure_ascii=False))
+        index = 0
+        with open(merge_file, "w", encoding="utf-8") as writer:
+            writer.write("#version: 0.2\n")
+            for bpe_tokens, token_index in sorted(
+                self.bpe_ranks.items(), key=lambda kv: kv[1]
+            ):
+                if index != token_index:
+                    logger.warning(
+                        "Saving vocabulary to {}: BPE merge indices are not consecutive."
+                        " Please check that the tokenizer is not corrupted!".format(
+                            merge_file
+                        )
+                    )
+                    index = token_index
+                writer.write(" ".join(bpe_tokens) + "\n")
+                index += 1
+        index = len(self.encoder)
+        with open(special_tokens_file, "w", encoding="utf-8") as writer:
+            for token, token_index in sorted(
+                self.special_tokens.items(), key=lambda kv: kv[1]
+            ):
+                if index != token_index:
+                    logger.warning(
+                        "Saving special tokens vocabulary to {}: BPE indices are not consecutive."
+                        " Please check that the tokenizer is not corrupted!".format(
+                            special_tokens_file
+                        )
+                    )
+                    index = token_index
+                writer.write(token + "\n")
+                index += 1
+        return vocab_file, merge_file, special_tokens_file

vocab/gpt_neox_chinese_v1/tokenizer/tokenizer.py ADDED Viewed

	@@ -0,0 +1,402 @@

+# Copyright (c) 2021, EleutherAI
+# This file is based on code by the authors denoted below and has been modified from its original version.
+#
+# Copyright (c) 2020, NVIDIA CORPORATION.  All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""Megatron tokenizers."""
+from abc import ABC
+from abc import abstractmethod
+from tokenizers import Tokenizer
+from transformers import GPT2Tokenizer, GPT2TokenizerFast
+import numpy as np
+import sentencepiece as spm
+from typing import List, Union
+from .gpt2_tokenization import GPT2Tokenizer
+def build_tokenizer(args):
+    """Initialize tokenizer."""
+    if args.rank == 0:
+        print("> building {} tokenizer ...".format(args.tokenizer_type), flush=True)
+    # Select and instantiate the tokenizer.
+    if args.tokenizer_type.lower() == "GPT2BPETokenizer".lower():
+        assert args.vocab_file is not None
+        assert args.merge_file is not None
+        tokenizer = _GPT2BPETokenizer(args.vocab_file, args.merge_file)
+    elif args.tokenizer_type.lower() == "SPMTokenizer".lower():
+        assert args.vocab_file is not None
+        tokenizer = SentencePieceTokenizer(args.vocab_file)
+    elif args.tokenizer_type.lower() == "HFTokenizer".lower():
+        assert args.vocab_file is not None
+        tokenizer = HFTokenizer(args.vocab_file)
+    elif args.tokenizer_type.lower() == "HFGPT2Tokenizer".lower():
+        if args.vocab_file is None:
+            print(
+                "WARNING: No vocab file found, loading Huggingface's pretrained GPT2Tokenizer"
+            )
+        tokenizer = HFGPT2Tokenizer(args.vocab_file)
+    elif args.tokenizer_type.lower() == "CharLevelTokenizer".lower():
+        tokenizer = CharLevelTokenizer(vocab_size=512)
+    elif args.tokenizer_type.lower() == "TiktokenTokenizer".lower():
+        assert args.vocab_file is not None
+        tokenizer = TiktokenTokenizer(args.vocab_file)
+    else:
+        raise NotImplementedError(
+            "{} tokenizer is not " "implemented.".format(args.tokenizer_type)
+        )
+    # Add vocab size.
+    args.padded_vocab_size = _vocab_size_with_padding(tokenizer.vocab_size, args)
+    return tokenizer
+def _vocab_size_with_padding(orig_vocab_size, args):
+    """Pad vocab size so it is divisible by model parallel size and
+    still having GPU friendly size."""
+    after = orig_vocab_size
+    multiple = args.make_vocab_size_divisible_by * args.model_parallel_size
+    while (after % multiple) != 0:
+        after += 1
+    if args.rank == 0:
+        print(
+            " > padded vocab (size: {}) with {} dummy tokens "
+            "(new size: {})".format(orig_vocab_size, after - orig_vocab_size, after),
+            flush=True,
+        )
+    return after
+class AbstractTokenizer(ABC):
+    """Abstract class for tokenizer."""
+    def __init__(self, name):
+        self.name = name
+        super().__init__()
+    @property
+    @abstractmethod
+    def vocab_size(self):
+        pass
+    @property
+    @abstractmethod
+    def vocab(self):
+        """Dictionary from vocab text token to id token."""
+        pass
+    @property
+    @abstractmethod
+    def inv_vocab(self):
+        """Dictionary from vocab id token to text token."""
+        pass
+    @abstractmethod
+    def tokenize(self, text):
+        pass
+    def detokenize(self, token_ids):
+        raise NotImplementedError(
+            "detokenizer is not implemented for {} " "tokenizer".format(self.name)
+        )
+    @property
+    def cls(self):
+        raise NotImplementedError(
+            "CLS is not provided for {} " "tokenizer".format(self.name)
+        )
+    @property
+    def sep(self):
+        raise NotImplementedError(
+            "SEP is not provided for {} " "tokenizer".format(self.name)
+        )
+    @property
+    def pad(self):
+        raise NotImplementedError(
+            "PAD is not provided for {} " "tokenizer".format(self.name)
+        )
+    @property
+    def eod(self):
+        raise NotImplementedError(
+            "EOD is not provided for {} " "tokenizer".format(self.name)
+        )
+    @property
+    def mask(self):
+        raise NotImplementedError(
+            "MASK is not provided for {} " "tokenizer".format(self.name)
+        )
+class _GPT2BPETokenizer(AbstractTokenizer):
+    """Original GPT2 BPE tokenizer."""
+    def __init__(self, vocab_file, merge_file):
+        name = "GPT2 BPE"
+        super().__init__(name)
+        self.tokenizer = GPT2Tokenizer(
+            vocab_file, merge_file, errors="replace", special_tokens=[], max_len=None
+        )
+        self.eod_id = self.tokenizer.encoder["<|endoftext|>"]
+    @property
+    def vocab_size(self):
+        return len(self.tokenizer.encoder)
+    @property
+    def vocab(self):
+        return self.tokenizer.encoder
+    @property
+    def inv_vocab(self):
+        return self.tokenizer.decoder
+    def tokenize(self, text):
+        return self.tokenizer.encode(text)
+    def detokenize(self, token_ids):
+        return self.tokenizer.decode(token_ids)
+    @property
+    def eod(self):
+        return self.eod_id
+class SentencePieceTokenizer(AbstractTokenizer):
+    """Designed to Integrate SP's Tokenizer."""
+    def __init__(self, vocab_file):
+        name = "SPM"
+        super().__init__(name)
+        self.tokenizer = spm.SentencePieceProcessor(model_file=vocab_file)
+        self.eod_id = self.tokenizer.piece_to_id("<|endoftext|>")
+    @property
+    def vocab_size(self):
+        return self.tokenizer.get_piece_size()
+    @property
+    def vocab(self):
+        return {
+            self.tokenizer.id_to_piece(idx): idx
+            for idx in range(self.tokenizer.get_piece_size())
+        }
+    @property
+    def inv_vocab(self):
+        return {
+            idx: self.tokenizer.id_to_piece(idx)
+            for idx in range(self.tokenizer.get_piece_size())
+        }
+    def tokenize(self, text):
+        return self.tokenizer.encode(text)
+    def detokenize(self, token_ids):
+        return self.tokenizer.decode(token_ids)
+    @property
+    def eod(self):
+        return self.eod_id
+class HFTokenizer(AbstractTokenizer):
+    """Designed to Integrate HF's Tokenizer library."""
+    def __init__(self, vocab_file):
+        name = "HFTokenizer"
+        super().__init__(name)
+        self.tokenizer = Tokenizer.from_file(vocab_file)
+        self.eod_id = self.tokenizer.token_to_id("<|endoftext|>")
+        self.pad_id = self.tokenizer.token_to_id("<|padding|>")
+    @property
+    def vocab_size(self):
+        return self.tokenizer.get_vocab_size()
+    @property
+    def vocab(self):
+        return self.tokenizer.get_vocab()
+    @property
+    def inv_vocab(self):
+        return self.tokenizer.decoder
+    def tokenize(self, text: str):
+        return self.tokenizer.encode(text).ids
+    def tokenize_batch(self, text_batch: Union[List[str], str]):
+        return self.tokenizer.encode_batch(text_batch)
+    def detokenize(self, token_ids):
+        return self.tokenizer.decode(token_ids)
+    @property
+    def eod(self):
+        return self.eod_id
+class HFGPT2Tokenizer(AbstractTokenizer):
+    """Designed to Integrate the pretrained OpenAI GPT2 Tokenizers from HF"""
+    def __init__(self, vocab_file=None, fast=True):
+        name = "HFGPT2Tokenizer"
+        if fast:
+            name += "Fast"
+        super().__init__(name)
+        if vocab_file is None:
+            vocab_file = "gpt2"
+        if fast:
+            self.tokenizer = GPT2TokenizerFast.from_pretrained(vocab_file)
+        else:
+            self.tokenizer = GPT2Tokenizer.from_pretrained(vocab_file)
+        self.tokenizer.add_special_tokens({"pad_token": "<|padding|>"})
+        self.eod_id = self.tokenizer.eos_token_id
+        self.pad_id = self.tokenizer.pad_token_id
+    @property
+    def vocab_size(self):
+        return len(self.tokenizer)
+    @property
+    def vocab(self):
+        return self.tokenizer.get_vocab()
+    @property
+    def inv_vocab(self):
+        return self.tokenizer._tokenizer.decoder
+    def tokenize(self, text: str):
+        return self.tokenizer.encode(text)
+    def tokenize_batch(self, text_batch: Union[List[str], str]):
+        if isinstance(text_batch, str):
+            text_batch = [text_batch]
+        return [self.tokenize(t) for t in text_batch]
+    def detokenize(self, token_ids):
+        return self.tokenizer.decode(token_ids)
+    @property
+    def eod(self):
+        return self.eod_id
+class CharLevelTokenizer(AbstractTokenizer):
+    """Character Level Tokenizer"""
+    def __init__(self, vocab_size):
+        name = "CharLevelTokenizer"
+        super().__init__(name)
+        self._vocab_size = vocab_size
+        self.eod_id = 0
+        self.pad_id = 1
+    def clamp(self, n):
+        return max(32, min(n, self.vocab_size))
+    @property
+    def vocab_size(self):
+        return self._vocab_size
+    @property
+    def vocab(self):
+        raise NotImplementedError
+    @property
+    def inv_vocab(self):
+        raise NotImplementedError
+    def decode_token(self, token: int):
+        return str(chr(self.clamp(token)))
+    def tokenize(self, text: str):
+        return list(np.fromstring(text, dtype=np.uint8))
+    def tokenize_batch(self, text_batch: Union[List[str], str]):
+        if isinstance(text_batch, list):
+            return [self.tokenize(s) for s in text_batch]
+        else:
+            return self.tokenize(text_batch)
+    def detokenize(self, token_ids):
+        return "".join(list(map(self.decode_token, token_ids)))
+    @property
+    def eod(self):
+        return self.eod_id
+class TiktokenTokenizer(AbstractTokenizer):
+    """Tokenizer from OpenAI's tiktoken implementation"""
+    def __init__(self, vocab_file):
+        try:
+            import tiktoken
+        except ModuleNotFoundError:
+            print("Please install tiktoken: (https://github.com/openai/tiktoken)")
+            raise Exception
+        name = "TiktokenTokenizer"
+        super().__init__(name)
+        self.tokenizer = tiktoken.get_encoding(vocab_file)
+        self.eod_id = self.tokenizer.eot_token
+        self.pad_id = None
+    @property
+    def vocab_size(self):
+        return self.tokenizer.n_vocab
+    @property
+    def vocab(self):
+        raise NotImplementedError(
+            "TiktokenTokenizer does not implement vocabulary access."
+        )
+    @property
+    def inv_vocab(self):
+        raise NotImplementedError(
+            "TiktokenTokenizer does not implement vocabulary access. \
+                To get the idx-th token in vocabulary, use tokenizer.decode([idx]) ."
+        )
+    def tokenize(self, text: str):
+        return self.tokenizer.encode(text)  # ,  allowed_special="all")
+    def tokenize_batch(self, text_batch: List[str]):
+        return self.tokenizer.encode_batch(text_batch, allowed_special="all")
+    def detokenize(self, token_ids):
+        return self.tokenizer.decode(tokens=token_ids, errors="strict")
+    @property
+    def eod(self):
+        return self.eod_id
+    @property
+    def pad(self):
+        raise NotImplementedError

vocab/gpt_neox_chinese_v1/tokenizer/train_tokenizer.py ADDED Viewed

	@@ -0,0 +1,126 @@

+# Copyright (c) 2021, EleutherAI
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+"""
+Assumes a dataset of jsonl files in the same format as the neox training set.
+"""
+from tokenizers import Tokenizer, decoders, models, pre_tokenizers, processors, trainers
+from tokenizers.normalizers import NFKC
+from glob import glob
+import os
+import json
+import argparse
+def load_jsonl(input_path, quiet=True) -> list:
+    """
+    Read list of objects from a JSON lines file.
+    """
+    data = []
+    with open(input_path, "r", encoding="utf-8") as f:
+        for line in f:
+            data.append(json.loads(line.rstrip("\n|\r")))
+    if not quiet:
+        print("Loaded {} records from {}".format(len(data), input_path))
+    return data
+def json_iterator(input_dir, text_key="text"):
+    all_jsonls = glob(f"{input_dir}/*.jsonl") + glob(f"{input_dir}/*.json")
+    for j in all_jsonls:
+        data = load_jsonl(j)
+        for doc in data:
+            yield doc[text_key]
+def train_tokenizer(
+    input_dir: str, save_path: str, tokenizer_type: str = "BPE", vocab_size: int = 52000
+):
+    """
+    Trains a tokenizer on all the json files in `input_dir` and saves it to `save_path`
+    :param input_dir: input directory containing jsonl files
+    :param save_path: path to save tokenizer to
+    :param tokenizer_type: type of tokenizer to train.
+    :param vocab_size: int, size of tokenizer's vocab
+    :return:
+    """
+    if tokenizer_type == "BPE":
+        model = models.BPE()
+    else:
+        raise NotImplementedError(f"Tokenizer type {tokenizer_type} not implemented")
+    tokenizer = Tokenizer(model)
+    # Customize pre-tokenization and decoding
+    tokenizer.pre_tokenizer = pre_tokenizers.ByteLevel(add_prefix_space=True)
+    tokenizer.decoder = decoders.ByteLevel()
+    tokenizer.post_processor = processors.ByteLevel(trim_offsets=True)
+    tokenizer.normalizer = NFKC()
+    # And then train
+    trainer = trainers.BpeTrainer(
+        vocab_size=vocab_size, special_tokens=["<|endoftext|>", "<|padding|>"]
+    )
+    tokenizer.train_from_iterator(json_iterator(input_dir), trainer)
+    # And Save it
+    tokenizer.save(save_path, pretty=True)
+    print(f"Tokenizer saved at {save_path}")
+def parse_args():
+    parser = argparse.ArgumentParser(
+        description="script for training a multilingual "
+        "HF tokenizer on CC dumps with upweighting for low resource languages"
+    )
+    parser.add_argument(
+        "--json_input_dir",
+        type=str,
+        help="Path to folder containing tokenizer training data in jsonl format",
+    )
+    parser.add_argument(
+        "--tokenizer_output_path",
+        type=str,
+        help="Path to which your trained tokenizer will be saved (should end in .json)",
+    )
+    parser.add_argument(
+        "--tokenizer_type",
+        type=str,
+        help="type of tokenizer to train, currently only BPE is supported",
+        choices=["BPE"],
+        default=["BPE"],
+    )
+    parser.add_argument(
+        "-v",
+        "--vocab_size",
+        help="vocabulary size of tokenizer, default=52k",
+        type=int,
+        default=52000,
+    )
+    return parser.parse_args()
+if __name__ == "__main__":
+    args = parse_args()
+    train_tokenizer(
+        args.json_input_dir,
+        save_path=args.tokenizer_output_path,
+        tokenizer_type=args.tokenizer_type,
+        vocab_size=args.vocab_size,
+    )