hereticness/Heretic-OpenCoder-1.5B-Instruct
122
1# coding=utf-82# Copyright 2022 EleutherAI and the HuggingFace Inc. team. All rights reserved.3#4# This code is based on EleutherAI's GPT-NeoX library and the GPT-NeoX5# and OPT implementations in this library. It has been modified from its6# original forms to accommodate minor architectural differences compared7# to GPT-NeoX and OPT used by the Meta AI team that trained the model.8#9# Licensed under the Apache License, Version 2.0 (the "License");10# you may not use this file except in compliance with the License.11# You may obtain a copy of the License at12#13# http://www.apache.org/licenses/LICENSE-2.014#15# Unless required by applicable law or agreed to in writing, software16# distributed under the License is distributed on an "AS IS" BASIS,17# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.18# See the License for the specific language governing permissions and19# limitations under the License.20 21"""Tokenization classes for INFLMTokenizer."""22import os23from shutil import copyfile24from typing import Any, Dict, List, Optional, Tuple25 26import sentencepiece as spm27 28from transformers.tokenization_utils import PreTrainedTokenizer29from transformers.utils import logging30 31from tokenizers import pre_tokenizers,Regex,decoders32from tokenizers.pre_tokenizers import Digits, Split, ByteLevel33import os 34 35# same as gpt4 cl-base-100k36PATTERN = Regex("(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?\p{L}+|\p{N}{1,3}| ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+\s+(\S)+")37 38logger = logging.get_logger(__name__)39 40VOCAB_FILES_NAMES = {"vocab_file": "./tokenizer.model"}41 42PRETRAINED_VOCAB_FILES_MAP = {}43 44 45class INFLMTokenizer(PreTrainedTokenizer):46 """47 Construct a INFLMTokenizer tokenizer based on sentence-piece 48 49 Args:50 vocab_file (`str`):51 Path to the vocabulary file.52 """53 54 vocab_files_names = VOCAB_FILES_NAMES55 pretrained_vocab_files_map = PRETRAINED_VOCAB_FILES_MAP56 model_input_names = ["input_ids", "attention_mask"]57 _auto_class = "AutoTokenizer"58 59 def __init__(60 self,61 vocab_file,62 unk_token="<unk>",63 bos_token="<s>",64 eos_token="</s>",65 pad_token="<pad>",66 sp_model_kwargs: Optional[Dict[str, Any]] = None,67 add_bos_token=False,68 add_eos_token=False,69 decode_with_prefix_space=False,70 clean_up_tokenization_spaces=False,71 spaces_between_special_tokens=False,72 **kwargs,73 ):74 self.sp_model_kwargs = {} if sp_model_kwargs is None else sp_model_kwargs75 self.vocab_file = vocab_file76 self.add_bos_token = add_bos_token77 self.add_eos_token = add_eos_token78 self.decode_with_prefix_space = decode_with_prefix_space79 self.sp_model = spm.SentencePieceProcessor(**self.sp_model_kwargs)80 self.sp_model.Load(vocab_file)81 self._no_prefix_space_tokens = None82 self.pre_tokenizer = pre_tokenizers.Sequence([Split(pattern =PATTERN,behavior = "isolated", invert = False)])83 super().__init__(84 bos_token=bos_token,85 eos_token=eos_token,86 unk_token=unk_token,87 pad_token=pad_token,88 clean_up_tokenization_spaces=clean_up_tokenization_spaces,89 spaces_between_special_tokens=spaces_between_special_tokens,90 **kwargs,91 ) 92 93 """ Initialisation"""94 95 @property96 def no_prefix_space_tokens(self):97 if self._no_prefix_space_tokens is None:98 vocab = self.convert_ids_to_tokens(list(range(self.vocab_size)))99 self._no_prefix_space_tokens = {i for i, tok in enumerate(vocab) if not tok.startswith("▁")}100 return self._no_prefix_space_tokens101 102 @property103 def vocab_size(self):104 """Returns vocab size"""105 return self.sp_model.get_piece_size()106 107 @property108 def bos_token_id(self) -> Optional[int]:109 return self.sp_model.bos_id()110 111 @property112 def eos_token_id(self) -> Optional[int]:113 return self.sp_model.eos_id()114 115 def get_vocab(self):116 """Returns vocab as a dict"""117 vocab = {self.convert_ids_to_tokens(i): i for i in range(self.vocab_size)}118 vocab.update(self.added_tokens_encoder)119 return vocab120 121 def _tokenize(self, text):122 """Returns a tokenized string."""123 124 splits = self.pre_tokenizer.pre_tokenize_str(text)125 texts=[]126 127 for split in splits:128 texts.extend(self.sp_model.encode(split[0], out_type=str))129 return texts130 131 def _convert_token_to_id(self, token):132 """Converts a token (str) in an id using the vocab."""133 134 return self.sp_model.piece_to_id(token)135 136 def _convert_id_to_token(self, index):137 """Converts an index (integer) in a token (str) using the vocab."""138 token = self.sp_model.IdToPiece(index)139 return token140 141 def _maybe_add_prefix_space(self, tokens, decoded):142 if tokens and tokens[0] not in self.no_prefix_space_tokens:143 return " " + decoded144 else:145 return decoded146 147 def convert_tokens_to_string(self, tokens):148 """Converts a sequence of tokens (string) in a single string."""149 current_sub_tokens = []150 out_string = ""151 prev_is_special = False152 for token in tokens:153 # make sure that special tokens are not decoded using sentencepiece model154 if token in self.all_special_tokens:155 out_string += self.sp_model.decode(current_sub_tokens) + token156 prev_is_special = True157 current_sub_tokens = []158 else:159 current_sub_tokens.append(token)160 prev_is_special = False161 out_string += self.sp_model.decode(current_sub_tokens)162 163 return out_string164 165 def save_vocabulary(self, save_directory, filename_prefix: Optional[str] = None) -> Tuple[str]:166 """167 Save the vocabulary and special tokens file to a directory.168 169 Args:170 save_directory (`str`):171 The directory in which to save the vocabulary.172 173 Returns:174 `Tuple(str)`: Paths to the files saved.175 """176 if not os.path.isdir(save_directory):177 logger.error(f"Vocabulary path ({save_directory}) should be a directory")178 return179 out_vocab_file = os.path.join(180 save_directory, (filename_prefix + "-" if filename_prefix else "") + VOCAB_FILES_NAMES["vocab_file"]181 )182 183 if os.path.abspath(self.vocab_file) != os.path.abspath(out_vocab_file) and os.path.isfile(self.vocab_file):184 copyfile(self.vocab_file, out_vocab_file)185 elif not os.path.isfile(self.vocab_file):186 with open(out_vocab_file, "wb") as fi:187 content_spiece_model = self.sp_model.serialized_model_proto()188 fi.write(content_spiece_model)189 190 return (out_vocab_file,)191 192 def build_inputs_with_special_tokens(self, token_ids_0, token_ids_1=None):193 if self.add_bos_token:194 bos_token_ids = [self.bos_token_id]195 else:196 bos_token_ids = []197 198 output = bos_token_ids + token_ids_0199 200 if token_ids_1 is not None:201 output = output + token_ids_1202 203 if self.add_eos_token:204 output = output + [self.eos_token_id]205 206 return output207 208 def get_special_tokens_mask(209 self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None, already_has_special_tokens: bool = False210 ) -> List[int]:211 """212 Retrieve sequence ids from a token list that has no special tokens added. This method is called when adding213 special tokens using the tokenizer `prepare_for_model` method.214 215 Args:216 token_ids_0 (`List[int]`):217 List of IDs.218 token_ids_1 (`List[int]`, *optional*):219 Optional second list of IDs for sequence pairs.220 already_has_special_tokens (`bool`, *optional*, defaults to `False`):221 Whether or not the token list is already formatted with special tokens for the model.222 223 Returns:224 `List[int]`: A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.225 """226 if already_has_special_tokens:227 return super().get_special_tokens_mask(228 token_ids_0=token_ids_0, token_ids_1=token_ids_1, already_has_special_tokens=True229 )230 231 eos_token_id = [1] if self.add_eos_token else []232 if token_ids_1 is None:233 return ([0] * len(token_ids_0)) + eos_token_id234 return ([0] * len(token_ids_0)) + eos_token_id + ([0] * len(token_ids_1)) + eos_token_id235 236 237 def create_token_type_ids_from_sequences(238 self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None239 ) -> List[int]:240 """241 Creates a mask from the two sequences passed to be used in a sequence-pair classification task. An ALBERT242 sequence pair mask has the following format:243 244 ```245 0 0 0 0 0 0 0 0 0 0 0 1 1 1 1 1 1 1 1 1246 | first sequence | second sequence |247 ```248 249 if token_ids_1 is None, only returns the first portion of the mask (0s).250 251 Note this is only used for back compatiblity, thus list of zero is returned.252 253 Args:254 token_ids_0 (`List[int]`):255 List of ids.256 token_ids_1 (`List[int]`, *optional*):257 Optional second list of IDs for sequence pairs.258 259 Returns:260 `List[int]`: List of zeros.261 """262 eos = [self.eos_token_id]263 264 if token_ids_1 is None:265 return len(token_ids_0 + eos) * [0]266 return len(token_ids_0 + eos + token_ids_1 + eos) * [0]267 268 269 @property270 def default_chat_template(self):271 return None272 273 274 def decode(275 self,276 token_ids,277 skip_special_tokens: bool = False,278 clean_up_tokenization_spaces: Optional[bool] = False,279 spaces_between_special_tokens: bool = False,280 **kwargs,281 ) -> str:282 # default spaces_between_special_tokens should be false.283 if spaces_between_special_tokens:284 logger.warning_once('spaces_between_special_tokens is set. \285 It has no effect for bos,eos,pad,unk when transformers<=4.38.')286 return super().decode(287 token_ids,288 skip_special_tokens=skip_special_tokens,289 clean_up_tokenization_spaces=clean_up_tokenization_spaces,290 spaces_between_special_tokens=spaces_between_special_tokens,291 **kwargs,292 )