THUDM
/

glm-4-9b-chat-1m

Model card Files Files and versions Community

zR commited on Jun 20, 2024

Commit

b97dd95

·

1 Parent(s): b36cb68

fix with convert_tokens_to_string

Files changed (1) hide show

tokenization_chatglm.py +11 -4

tokenization_chatglm.py CHANGED Viewed

@@ -63,22 +63,22 @@ class ChatGLM4Tokenizer(PreTrainedTokenizer):
         vocab.update(self.added_tokens_encoder)
         return vocab
-    def convert_tokens_to_string(self, tokens: List[Union[bytes, str]]) -> str:
         """
         Converts a sequence of tokens in a single string.
         """
         text = ""
         temp = b""
         for t in tokens:
             if isinstance(t, str):
                 if temp:
                     text += temp.decode("utf-8", errors="replace")
-                    temp = b""
-                text += t
             elif isinstance(t, bytes):
                 temp += t
             else:
-                raise TypeError("token should only be of type types or str")
         if temp:
             text += temp.decode("utf-8", errors="replace")
         return text
@@ -90,6 +90,13 @@ class ChatGLM4Tokenizer(PreTrainedTokenizer):
             tokens.append(self.decoder[t])
         return tokens
     def _convert_token_to_id(self, token):
         """ Converts a token (str) in an id using the vocab. """
         return self.mergeable_ranks[token]

         vocab.update(self.added_tokens_encoder)
         return vocab
+    def convert_tokens_to_string(self, tokens: List[Union[bytes, str, int]]) -> str:
         """
         Converts a sequence of tokens in a single string.
         """
         text = ""
         temp = b""
         for t in tokens:
+            if isinstance(t, int):
+                t = chr(t)
             if isinstance(t, str):
                 if temp:
                     text += temp.decode("utf-8", errors="replace")
             elif isinstance(t, bytes):
                 temp += t
             else:
+                raise TypeError("token should only be of type int, bytes or str")
         if temp:
             text += temp.decode("utf-8", errors="replace")
         return text
             tokens.append(self.decoder[t])
         return tokens
+    def _tokenize(self, text, **kwargs):
+        tokens = []
+        ids = self.tokenizer.encode(text)
+        for t in ids:
+            tokens.append(self.decoder[t])
+        return tokens
     def _convert_token_to_id(self, token):
         """ Converts a token (str) in an id using the vocab. """
         return self.mergeable_ranks[token]