From 3226b3c5ef93998f5c412fb926ff136b50816bd7 Mon Sep 17 00:00:00 2001 From: Jonathan Soma Date: Tue, 30 Apr 2024 14:33:23 -0400 Subject: [PATCH] fix: UTF-8 handling with grammars (#1415) Use Python's built-in UTF-8 handling to get code points --- llama_cpp/llama_grammar.py | 16 +++++----------- 1 file changed, 5 insertions(+), 11 deletions(-) diff --git a/llama_cpp/llama_grammar.py b/llama_cpp/llama_grammar.py index 6c7b57a..d9a3823 100644 --- a/llama_cpp/llama_grammar.py +++ b/llama_cpp/llama_grammar.py @@ -556,17 +556,11 @@ def add_rule( # } def decode_utf8(src: const_char_p) -> Tuple[int, const_char_p]: """Decodes a UTF-8 character from the source string.""" - lookup = (1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 2, 3, 4) - first_byte = ord(src[0]) # type: int - highbits = first_byte >> 4 # type: int - len = lookup[highbits] # type: int - mask = (1 << (8 - len)) - 1 # type: int - value = first_byte & mask # type: int - end = src + len # type: const_char_p # may overrun! - pos = src + 1 # type: const_char_p - while pos < end and pos[0]: - value = (value << 6) + (ord(pos[0]) & 0x3F) - pos += 1 + # Get the codepoint of the first character + value = ord(src[0]) + # Move the pointer ahead one character + pos = src + 1 + return value, pos