CoolFace
Modelpublic

SystemCSer/SparkleLLM_v1

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
4likes
tokenization_chatglm.py382 linesDownload Raw Back to root
1import regex as re2import base643import os4import json5import tiktoken6from torch import TensorType7from typing import List, Optional, Union, Dict, Any8from transformers import PreTrainedTokenizer9from transformers.utils import logging, PaddingStrategy10from transformers.tokenization_utils_base import EncodedInput, BatchEncoding11 12 13class ChatGLM4Tokenizer(PreTrainedTokenizer):14    vocab_files_names = {"vocab_file": "tokenizer.model"}15    model_input_names = ["input_ids", "attention_mask", "position_ids"]16 17    def __init__(18            self,19            vocab_file,20            padding_side="left",21            clean_up_tokenization_spaces=False,22            encode_special_tokens=False,23            **kwargs24    ):25        self.name = "GLM4Tokenizer"26        self.vocab_file = vocab_file27        pat_str = "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?\\p{L}+|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+"28        self.pat_str = re.compile(pat_str)29        self.encode_special_tokens = encode_special_tokens30 31        mergeable_ranks = {}32        with open(vocab_file) as f:33            for line in f:34                token, rank = line.strip().split()35                rank = int(rank)36                token = base64.b64decode(token)37                mergeable_ranks[token] = rank38 39        self.mergeable_ranks = mergeable_ranks40 41        self.tokenizer = tiktoken.Encoding(42            name="my_tokenizer",43            pat_str=pat_str,44            mergeable_ranks=mergeable_ranks,45            special_tokens={}46        )47        self.decoder = {rank: token for token, rank in mergeable_ranks.items()}48        self.n_words = len(self.decoder)49 50        super().__init__(51            padding_side=padding_side,52            clean_up_tokenization_spaces=clean_up_tokenization_spaces,53            **kwargs54        )55 56    @property57    def vocab_size(self):58        return self.n_words59 60    def get_vocab(self):61        """ Returns vocab as a dict """62        vocab = {self._convert_id_to_token(i): i for i in range(self.vocab_size)}63        vocab.update(self.added_tokens_encoder)64        return vocab65 66    def convert_tokens_to_string(self, tokens: List[Union[bytes, str]]) -> str:67        """68        Converts a sequence of tokens in a single string.69        """70        text = ""71        temp = b""72        for t in tokens:73            if isinstance(t, str):74                if temp:75                    text += temp.decode("utf-8", errors="replace")76                    temp = b""77                text += t78            elif isinstance(t, bytes):79                temp += t80            else:81                raise TypeError("token should only be of type types or str")82        if temp:83            text += temp.decode("utf-8", errors="replace")84        return text85 86    def _tokenize(self, text, **kwargs):87        tokens = []88        ids = self.tokenizer.encode(text)89        for t in ids:90            tokens.append(self.decoder[t])91        return tokens92 93    def _convert_token_to_id(self, token):94        """ Converts a token (str) in an id using the vocab. """95        return self.mergeable_ranks[token]96 97    def _convert_id_to_token(self, index):98        """Converts an index (integer) in a token (str) using the vocab."""99        return self.decoder.get(index, "")100 101    def save_vocabulary(self, save_directory, filename_prefix=None):102        """103        Save the vocabulary and special tokens file to a directory.104 105        Args:106            save_directory (`str`):107                The directory in which to save the vocabulary.108            filename_prefix (`str`, *optional*):109                An optional prefix to add to the named of the saved files.110 111        Returns:112            `Tuple(str)`: Paths to the files saved.113        """114        if os.path.isdir(save_directory):115            vocab_file = os.path.join(116                save_directory, self.vocab_files_names["vocab_file"]117            )118        else:119            vocab_file = save_directory120 121        with open(self.vocab_file, 'rb') as fin:122            proto_str = fin.read()123 124        with open(vocab_file, "wb") as writer:125            writer.write(proto_str)126 127        return (vocab_file,)128 129    def get_prefix_tokens(self):130        prefix_tokens = [self.convert_tokens_to_ids("[gMASK]"), self.convert_tokens_to_ids("<sop>")]131        return prefix_tokens132 133    def build_single_message(self, role, metadata, message, tokenize=True):134        assert role in ["system", "user", "assistant", "observation"], role135        if tokenize:136            role_tokens = [self.convert_tokens_to_ids(f"<|{role}|>")] + self.tokenizer.encode(f"{metadata}\n",137                                                                                              disallowed_special=())138            message_tokens = self.tokenizer.encode(message, disallowed_special=())139            tokens = role_tokens + message_tokens140            return tokens141        else:142            return str(f"<|{role}|>{metadata}\n{message}")143 144    # Use Jinja Template in tokenizer_config.json145    # def apply_chat_template(146    #         self,147    #         conversation: Union[List[Dict[str, str]], List[List[Dict[str, str]]], "Conversation"],148    #         add_generation_prompt: bool = False,149    #         tokenize: bool = True,150    #         padding: bool = False,151    #         truncation: bool = False,152    #         max_length: Optional[int] = None,153    #         return_tensors: Optional[Union[str, TensorType]] = None,154    #         return_dict: bool = False,155    #         tokenizer_kwargs: Optional[Dict[str, Any]] = None,156    #         add_special_tokens: bool = True,157    #         **kwargs,158    # ) -> Union[str, List[int], List[str], List[List[int]], BatchEncoding]:159    #160    #     if return_dict and not tokenize:161    #         raise ValueError(162    #             "`return_dict=True` is incompatible with `tokenize=False`, because there is no dict "163    #             "of tokenizer outputs to return."164    #         )165    #166    #     def handle_single_conversation(conversation):167    #         input_ids = self.get_prefix_tokens() if add_special_tokens else []168    #         input_message = "[gMASK]<sop>" if add_special_tokens else ""169    #         for item in conversation:170    #             if item.get("tools"):171    #                 tools = item["tools"]172    #                 content = "你是一个名为 GhatGLM 的人工智能助手。你是基于智谱AI训练的语言模型 GLM-4 模型开发的,你的任务是针对用户的问题和要求提供适当的答复和支持。"173    #                 content += "\n\n# 可用工具"174    #                 for tool in tools:175    #                     if tool["type"] == "function":176    #                         function = tool["function"]177    #                         content += f"\n\n## {function['name']}\n\n{json.dumps(function, ensure_ascii=False, indent=4)}"178    #                         content += "\n在调用上述函数时,请使用 Json 格式表示调用的参数。"179    #                     elif tool["type"] == "python":180    #                         content += "\n\n## python\n\n当你向 `python` 发送包含 Python 代码的消息时,该代码将会在一个有状态的 Jupyter notebook 环境中执行。\n`python` 返回代码执行的输出,或在执行 60 秒后返回超时。\n`/mnt/data` 将会持久化存储你的文件。在此会话中,`python` 无法访问互联网。不要使用 `python` 进行任何网络请求或者在线 API 调用,这些在线内容的访问将不会成功。"181    #                     elif tool["type"] == "simple_browser":182    #                         content += "\n\n## simple_browser\n\n你可以使用 `simple_browser` 工具。该工具支持以下函数:\n`search(query: str, recency_days: int)`:使用搜索引擎进行查询并显示结果,可以使用 `recency_days` 参数控制搜索内容的时效性。\n`mclick(ids: list[int])`:获取一系列指定 id 的页面内容。每次调用时,须选择3-10个页面。选择多个角度的页面,同时尽可能选择可信任的信息来源。考虑到部分页面是无法加载的,你也可以多打开一些可能有用的页面而不用担心内容过多。\n`open_url(url: str)`:打开指定的 URL。\n\n使用 `【{引用 id}†{引用文本}】` 来引用内容。\n\n操作步骤:1. 使用 `search` 来获得信息列表; 2. 使用 `mclick` 来获取指定 ID 页面的内容; 3. 根据获得的内容进行回复。在回复中应当引用信息来源。\n 如果用户提供了 URL,也可以用 `open_url` 直接打开页面。\n如果初次搜索结果没有找到合适的信息,也可以再次使用 `search` 进行搜索。"183    #                     elif tool["type"] == "cogview":184    #                         content += "\n\n## cogview\n\n如果用户的请求中包含了对图像的描述,你可以使用 `cogview` 来生成图像并展示给用户。你需要向 `cogview` 发送图像描述,规则:\n- 发送给 `cogview` 的消息必须使用英语。用户的中文描述必须完全翻译为英语。\n- 应当尽可能详细地描述图像生成的需求,需求描述约 100 英文单词。\n- 保持用户原始描述的意图。不要虚构内容或者没见过的人物。\n- 如无特殊说明,所在地为中国,持有中国立场并遵循中国社会主义价值观。"185    #                     else:186    #                         raise NotImplementedError(f"Unknown tool type {tool['type']}")187    #                 input = self.build_single_message("system", "", content, tokenize=tokenize)188    #                 if tokenize:189    #                     input_ids.extend(input)190    #                 else:191    #                     input_message += input192    #             if item["content"]:193    #                 input = self.build_single_message(194    #                     item["role"],195    #                     item.get("metadata", ""),196    #                     item["content"],197    #                     tokenize=tokenize198    #                 )199    #                 if tokenize:200    #                     input_ids.extend(input)201    #                 else:202    #                     input_message += input203    #         if add_generation_prompt:204    #             if tokenize:205    #                 input_ids.extend([self.convert_tokens_to_ids("<|assistant|>")])206    #             else:207    #                 input_message += "<|assistant|>"208    #         return input_ids if tokenize else input_message209    #210    #     # Main logic to handle different conversation formats211    #     if isinstance(conversation, list) and all(isinstance(i, dict) for i in conversation):212    #         result = handle_single_conversation(conversation)213    #     elif isinstance(conversation, list) and all(isinstance(i, list) for i in conversation):214    #         result = [handle_single_conversation(c) for c in conversation]215    #     elif hasattr(conversation, "messages"):216    #         result = handle_single_conversation(conversation.messages)217    #     else:218    #         raise ValueError("Invalid conversation format")219    #220    #     if tokenize:221    #         output = self.batch_encode_plus(222    #             [result] if isinstance(result[0], int) else result,223    #             padding=padding,224    #             truncation=truncation,225    #             max_length=max_length,226    #             return_tensors=return_tensors,227    #             is_split_into_words=True,228    #             add_special_tokens=False229    #         )230    #         if return_dict:231    #             return output232    #         else:233    #             return output["input_ids"]234    #     else:235    #         return result236 237    def build_inputs_with_special_tokens(238            self, token_ids_0: List[int], token_ids_1: Optional[List[int]] = None239    ) -> List[int]:240        """241        Build model inputs from a sequence or a pair of sequence for sequence classification tasks by concatenating and242        adding special tokens. A BERT sequence has the following format:243 244        - single sequence: `[CLS] X [SEP]`245        - pair of sequences: `[CLS] A [SEP] B [SEP]`246 247        Args:248            token_ids_0 (`List[int]`):249                List of IDs to which the special tokens will be added.250            token_ids_1 (`List[int]`, *optional*):251                Optional second list of IDs for sequence pairs.252 253        Returns:254            `List[int]`: List of [input IDs](../glossary#input-ids) with the appropriate special tokens.255        """256        prefix_tokens = self.get_prefix_tokens()257        token_ids_0 = prefix_tokens + token_ids_0258        if token_ids_1 is not None:259            token_ids_0 = token_ids_0 + token_ids_1 + [self.convert_tokens_to_ids("<eos>")]260        return token_ids_0261 262    def _pad(263            self,264            encoded_inputs: Union[Dict[str, EncodedInput], BatchEncoding],265            max_length: Optional[int] = None,266            padding_strategy: PaddingStrategy = PaddingStrategy.DO_NOT_PAD,267            pad_to_multiple_of: Optional[int] = None,268            return_attention_mask: Optional[bool] = None,269    ) -> dict:270        """271        Pad encoded inputs (on left/right and up to predefined length or max length in the batch)272 273        Args:274            encoded_inputs:275                Dictionary of tokenized inputs (`List[int]`) or batch of tokenized inputs (`List[List[int]]`).276            max_length: maximum length of the returned list and optionally padding length (see below).277                Will truncate by taking into account the special tokens.278            padding_strategy: PaddingStrategy to use for padding.279 280                - PaddingStrategy.LONGEST Pad to the longest sequence in the batch281                - PaddingStrategy.MAX_LENGTH: Pad to the max length (default)282                - PaddingStrategy.DO_NOT_PAD: Do not pad283                The tokenizer padding sides are defined in self.padding_side:284 285                    - 'left': pads on the left of the sequences286                    - 'right': pads on the right of the sequences287            pad_to_multiple_of: (optional) Integer if set will pad the sequence to a multiple of the provided value.288                This is especially useful to enable the use of Tensor Core on NVIDIA hardware with compute capability289                `>= 7.5` (Volta).290            return_attention_mask:291                (optional) Set to False to avoid returning attention mask (default: set to model specifics)292        """293        # Load from model defaults294        assert self.padding_side == "left"295 296        required_input = encoded_inputs[self.model_input_names[0]]297        seq_length = len(required_input)298 299        if padding_strategy == PaddingStrategy.LONGEST:300            max_length = len(required_input)301 302        if max_length is not None and pad_to_multiple_of is not None and (max_length % pad_to_multiple_of != 0):303            max_length = ((max_length // pad_to_multiple_of) + 1) * pad_to_multiple_of304 305        needs_to_be_padded = padding_strategy != PaddingStrategy.DO_NOT_PAD and len(required_input) != max_length306 307        # Initialize attention mask if not present.308        if "attention_mask" not in encoded_inputs:309            encoded_inputs["attention_mask"] = [1] * seq_length310 311        if "position_ids" not in encoded_inputs:312            encoded_inputs["position_ids"] = list(range(seq_length))313 314        if needs_to_be_padded:315            difference = max_length - len(required_input)316 317            if "attention_mask" in encoded_inputs:318                encoded_inputs["attention_mask"] = [0] * difference + encoded_inputs["attention_mask"]319            if "position_ids" in encoded_inputs:320                encoded_inputs["position_ids"] = [0] * difference + encoded_inputs["position_ids"]321            encoded_inputs[self.model_input_names[0]] = [self.pad_token_id] * difference + required_input322 323        return encoded_inputs324 325    @property326    def default_chat_template(self):327        """328        GLM-4 uses [gMASK] and <sop> to indicate user messages. The system message is included as part of the first user329        message. The assistant messages do not have special tokens, as they can be identified by their order.330        """331        template = (332            "{% if messages[0]['role'] == 'system' %}"333            "{% set loop_messages = messages[1:] %}"  # Extract system message if it's present334            "{% set system_message = messages[0]['content'] %}"335            "{% elif USE_DEFAULT_PROMPT == true and not '[gMASK]' in messages[0]['content'] %}"336            "{% set loop_messages = messages %}"  # Or use the default system message if the flag is set337            "{% set system_message = 'DEFAULT_SYSTEM_MESSAGE' %}"338            "{% else %}"339            "{% set loop_messages = messages %}"340            "{% set system_message = false %}"341            "{% endif %}"342            "{% for message in loop_messages %}"  # Loop over all non-system messages343            "{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}"344            "{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}"345            "{% endif %}"346            "{% if loop.index0 == 0 and system_message != false %}"  # Embed system message in first message347            "{% set content = '[gMASK]<sop>' + system_message + '\\n' + message['content'] %}"348            "{% else %}"349            "{% set content = message['content'] %}"350            "{% endif %}"351            "{% if message['role'] == 'user' %}"  # Handle user messages352            "{{ content.strip() }}"353            "{% elif message['role'] == 'assistant' %}"  # Handle assistant messages354            "{{ ' '  + content.strip() + ' ' }}"355            "{% endif %}"356            "{% endfor %}"357            "{% if add_generation_prompt %}{% endif %}"358        )359        template = template.replace("USE_DEFAULT_PROMPT", "true" if self.use_default_system_prompt else "false")360        default_message = "你是一个名为 GLM-4 的人工智能助手。你是基于智谱AI训练的语言模型 GLM-4 模型开发的,你的任务是针对用户的问题和要求提供适当的答复和支持。\n\n# 可用工具\n"361        default_message += "\n## python\n\n当你向 `python` 发送包含 Python 代码的消息时,该代码将会在一个有状态的 Jupyter notebook 环境中执行。\n"362        default_message += "`python` 返回代码执行的输出,或在执行 60 秒后返回超时。\n`/mnt/data` 将会持久化存储你的文件。在此会话中,`python` 无法访问互联网。"363        default_message += "不要使用 `python` 进行任何网络请求或者在线 API 调用,这些在线内容的访问将不会成功。"364        default_message += "\n## simple_browser\n\n你可以使用 `simple_browser` 工具。该工具支持以下函数:\n"365        default_message += "`search(query: str, recency_days: int)`:使用搜索引擎进行查询并显示结果,可以使用 `recency_days` 参数控制搜索内容的时效性。\n"366        default_message += "`mclick(ids: list[int])`:获取一系列指定 id 的页面内容。每次调用时,须选择3-10个页面。选择多个角度的页面,同时尽可能选择可信任的信息来源。"367        default_message += "考虑到部分页面是无法加载的,你也可以多打开一些可能有用的页面而不用担心内容过多。\n"368        default_message += "`open_url(url: str)`:打开指定的 URL。\n"369        default_message += "使用 `【{引用 id}†{引用文本}】` 来引用内容。\n\n"370        default_message += "操作步骤:1. 使用 `search` 来获得信息列表; 2. 使用 `mclick` 来获取指定 ID 页面的内容; 3. 根据获得的内容进行回复。"371        default_message += "在回复中应当引用信息来源。\n 如果用户提供了 URL,也可以用 `open_url` 直接打开页面。"372        default_message += "\n如果初次搜索结果没有找到合适的信息,也可以再次使用 `search` 进行搜索。\n"373        default_message += "## cogview\n\n如果用户的请求中包含了对图像的描述,你可以使用 `cogview` 来生成图像并展示给用户。"374        default_message += "你需要向 `cogview` 发送图像描述,规则:\n- 发送给 `cogview` 的消息必须使用英语。用户的中文描述必须完全翻译为英语。\n"375        default_message += "- 应当尽可能详细地描述图像生成的需求,需求描述约 100 英文单词。\n- 保持用户原始描述的意图。不要虚构内容或者没见过的人物。\n"376        default_message += "- 如无特殊说明,所在地为中国,持有中国立场并遵循中国社会主义价值观。"377        default_message = default_message.replace("\n", "\\n").replace("'", "\\'")378        template = template.replace("DEFAULT_SYSTEM_MESSAGE", default_message)379 380        return template381 382