Prompt48 commited on
Commit
aea852c
·
verified ·
1 Parent(s): cb9f744

Upload edit\Qwen3-TTS-test\.venv\Lib\site-packages\transformers\models\herbert\tokenization_herbert_fast.py with huggingface_hub

Browse files
edit//Qwen3-TTS-test//.venv//Lib//site-packages//transformers//models//herbert//tokenization_herbert_fast.py ADDED
@@ -0,0 +1,133 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # coding=utf-8
2
+ # Copyright 2020 The Google AI Language Team Authors, Allegro.pl, Facebook Inc. and the HuggingFace Inc. team.
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+
16
+ from typing import Optional
17
+
18
+ from ...tokenization_utils_fast import PreTrainedTokenizerFast
19
+ from ...utils import logging
20
+ from .tokenization_herbert import HerbertTokenizer
21
+
22
+
23
+ logger = logging.get_logger(__name__)
24
+
25
+ VOCAB_FILES_NAMES = {"vocab_file": "vocab.json", "merges_file": "merges.txt", "tokenizer_file": "tokenizer.json"}
26
+
27
+
28
+ class HerbertTokenizerFast(PreTrainedTokenizerFast):
29
+ """
30
+ Construct a "Fast" BPE tokenizer for HerBERT (backed by HuggingFace's *tokenizers* library).
31
+
32
+ Peculiarities:
33
+
34
+ - uses BERT's pre-tokenizer: BertPreTokenizer splits tokens on spaces, and also on punctuation. Each occurrence of
35
+ a punctuation character will be treated separately.
36
+
37
+ This tokenizer inherits from [`PreTrainedTokenizer`] which contains most of the methods. Users should refer to the
38
+ superclass for more information regarding methods.
39
+
40
+ Args:
41
+ vocab_file (`str`):
42
+ Path to the vocabulary file.
43
+ merges_file (`str`):
44
+ Path to the merges file.
45
+ """
46
+
47
+ vocab_files_names = VOCAB_FILES_NAMES
48
+ slow_tokenizer_class = HerbertTokenizer
49
+
50
+ def __init__(
51
+ self,
52
+ vocab_file=None,
53
+ merges_file=None,
54
+ tokenizer_file=None,
55
+ cls_token="<s>",
56
+ unk_token="<unk>",
57
+ pad_token="<pad>",
58
+ mask_token="<mask>",
59
+ sep_token="</s>",
60
+ **kwargs,
61
+ ):
62
+ super().__init__(
63
+ vocab_file,
64
+ merges_file,
65
+ tokenizer_file=tokenizer_file,
66
+ cls_token=cls_token,
67
+ unk_token=unk_token,
68
+ pad_token=pad_token,
69
+ mask_token=mask_token,
70
+ sep_token=sep_token,
71
+ **kwargs,
72
+ )
73
+
74
+ def build_inputs_with_special_tokens(
75
+ self, token_ids_0: list[int], token_ids_1: Optional[list[int]] = None
76
+ ) -> list[int]:
77
+ """
78
+ Build model inputs from a sequence or a pair of sequence for sequence classification tasks by concatenating and
79
+ adding special tokens. An HerBERT, like BERT sequence has the following format:
80
+
81
+ - single sequence: `<s> X </s>`
82
+ - pair of sequences: `<s> A </s> B </s>`
83
+
84
+ Args:
85
+ token_ids_0 (`List[int]`):
86
+ List of IDs to which the special tokens will be added.
87
+ token_ids_1 (`List[int]`, *optional*):
88
+ Optional second list of IDs for sequence pairs.
89
+
90
+ Returns:
91
+ `List[int]`: List of [input IDs](../glossary#input-ids) with the appropriate special tokens.
92
+ """
93
+
94
+ cls = [self.cls_token_id]
95
+ sep = [self.sep_token_id]
96
+ if token_ids_1 is None:
97
+ return cls + token_ids_0 + sep
98
+
99
+ return cls + token_ids_0 + sep + token_ids_1 + sep
100
+
101
+ def get_special_tokens_mask(
102
+ self, token_ids_0: list[int], token_ids_1: Optional[list[int]] = None, already_has_special_tokens: bool = False
103
+ ) -> list[int]:
104
+ """
105
+ Retrieve sequence ids from a token list that has no special tokens added. This method is called when adding
106
+ special tokens using the tokenizer `prepare_for_model` method.
107
+
108
+ Args:
109
+ token_ids_0 (`List[int]`):
110
+ List of IDs.
111
+ token_ids_1 (`List[int]`, *optional*):
112
+ Optional second list of IDs for sequence pairs.
113
+ already_has_special_tokens (`bool`, *optional*, defaults to `False`):
114
+ Whether or not the token list is already formatted with special tokens for the model.
115
+
116
+ Returns:
117
+ `List[int]`: A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.
118
+ """
119
+ if already_has_special_tokens:
120
+ return super().get_special_tokens_mask(
121
+ token_ids_0=token_ids_0, token_ids_1=token_ids_1, already_has_special_tokens=True
122
+ )
123
+
124
+ if token_ids_1 is None:
125
+ return [1] + ([0] * len(token_ids_0)) + [1]
126
+ return [1] + ([0] * len(token_ids_0)) + [1] + ([0] * len(token_ids_1)) + [1]
127
+
128
+ def save_vocabulary(self, save_directory: str, filename_prefix: Optional[str] = None) -> tuple[str]:
129
+ files = self._tokenizer.model.save(save_directory, name=filename_prefix)
130
+ return tuple(files)
131
+
132
+
133
+ __all__ = ["HerbertTokenizerFast"]