llama : refactor unicode stuff (ggerganov#5992)

* llama : refactor unicode stuff ggml-ci * unicode : names * make : fix c++ compiler * unicode : names * unicode : straighten tables * zig : fix build * unicode : put nfd normalization behind API ggml-ci * swift : fix build * unicode : add BOM * unicode : add <cstdint> ggml-ci * unicode : pass as cpts as const ref
NeoZhangJianyu · Mar 12, 2024 · d0b01ef · d0b01ef
1 parent 0a8db3f
commit d0b01ef
Show file tree

Hide file tree

Showing 9 changed files with 1,744 additions and 836 deletions.
diff --git a/CMakeLists.txt b/CMakeLists.txt
@@ -1141,6 +1141,8 @@ endif()
 add_library(llama
  llama.cpp
  llama.h
+ unicode.h
+ unicode.cpp
  )
 
 target_include_directories(llama PUBLIC .)

diff --git a/Makefile b/Makefile
@@ -633,9 +633,12 @@ ggml-backend.o: ggml-backend.c ggml.h ggml-backend.h
 ggml-quants.o: ggml-quants.c ggml.h ggml-quants.h ggml-common.h
  $(CC) $(CFLAGS) -c $< -o $@
 
-OBJS += ggml-alloc.o ggml-backend.o ggml-quants.o
+unicode.o: unicode.cpp unicode.h
+ $(CXX) $(CXXFLAGS) -c $< -o $@
+
+OBJS += ggml-alloc.o ggml-backend.o ggml-quants.o unicode.o
 
-llama.o: llama.cpp ggml.h ggml-alloc.h ggml-backend.h ggml-cuda.h ggml-metal.h llama.h
+llama.o: llama.cpp unicode.h ggml.h ggml-alloc.h ggml-backend.h ggml-cuda.h ggml-metal.h llama.h
  $(CXX) $(CXXFLAGS) -c $< -o $@
 
 COMMON_H_DEPS = common/common.h common/sampling.h common/log.h

diff --git a/Package.swift b/Package.swift
@@ -31,6 +31,7 @@ let package = Package(
  sources: [
  "ggml.c",
  "llama.cpp",
+ "unicode.cpp",
  "ggml-alloc.c",
  "ggml-backend.c",
  "ggml-quants.c",

diff --git a/build.zig b/build.zig
@@ -115,6 +115,7 @@ pub fn build(b: *std.build.Builder) !void {
  const ggml_alloc = make.obj("ggml-alloc", "ggml-alloc.c");
  const ggml_backend = make.obj("ggml-backend", "ggml-backend.c");
  const ggml_quants = make.obj("ggml-quants", "ggml-quants.c");
+ const unicode = make.obj("unicode", "unicode.cpp");
  const llama = make.obj("llama", "llama.cpp");
  const buildinfo = make.obj("common", "common/build-info.cpp");
  const common = make.obj("common", "common/common.cpp");
@@ -125,14 +126,14 @@ pub fn build(b: *std.build.Builder) !void {
  const clip = make.obj("clip", "examples/llava/clip.cpp");
  const llava = make.obj("llava", "examples/llava/llava.cpp");
 
- _ = make.exe("main", "examples/main/main.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, common, buildinfo, sampling, console, grammar_parser });
- _ = make.exe("quantize", "examples/quantize/quantize.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, common, buildinfo });
- _ = make.exe("perplexity", "examples/perplexity/perplexity.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, common, buildinfo });
- _ = make.exe("embedding", "examples/embedding/embedding.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, common, buildinfo });
- _ = make.exe("finetune", "examples/finetune/finetune.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, common, buildinfo, train });
- _ = make.exe("train-text-from-scratch", "examples/train-text-from-scratch/train-text-from-scratch.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, common, buildinfo, train });
+ _ = make.exe("main", "examples/main/main.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, unicode, common, buildinfo, sampling, console, grammar_parser });
+ _ = make.exe("quantize", "examples/quantize/quantize.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, unicode, common, buildinfo });
+ _ = make.exe("perplexity", "examples/perplexity/perplexity.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, unicode, common, buildinfo });
+ _ = make.exe("embedding", "examples/embedding/embedding.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, unicode, common, buildinfo });
+ _ = make.exe("finetune", "examples/finetune/finetune.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, unicode, common, buildinfo, train });
+ _ = make.exe("train-text-from-scratch", "examples/train-text-from-scratch/train-text-from-scratch.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, unicode, common, buildinfo, train });
 
- const server = make.exe("server", "examples/server/server.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, common, buildinfo, sampling, grammar_parser, clip, llava });
+ const server = make.exe("server", "examples/server/server.cpp", &.{ ggml, ggml_alloc, ggml_backend, ggml_quants, llama, unicode, common, buildinfo, sampling, grammar_parser, clip, llava });
  if (server.target.isWindows()) {
  server.linkSystemLibrary("ws2_32");
  }

diff --git a/llama.cpp b/llama.cpp
@@ -3703,7 +3703,7 @@ static void llm_load_vocab(
 
  for (int i = 0; i < n_merges; i++) {
  const std::string word = gguf_get_arr_str(ctx, merges_keyidx, i);
- GGML_ASSERT(codepoints_from_utf8(word).size() > 0);
+ GGML_ASSERT(unicode_cpts_from_utf8(word).size() > 0);
 
  std::string first;
  std::string second;
@@ -3748,7 +3748,7 @@ static void llm_load_vocab(
 
  for (uint32_t i = 0; i < n_vocab; i++) {
  std::string word = gguf_get_arr_str(ctx, token_idx, i);
- GGML_ASSERT(codepoints_from_utf8(word).size() > 0);
+ GGML_ASSERT(unicode_cpts_from_utf8(word).size() > 0);
 
  vocab.token_to_id[word] = i;
 
@@ -9350,7 +9350,7 @@ static uint8_t llama_token_to_byte(const llama_vocab& vocab, llama_token id) {
  }
  case LLAMA_VOCAB_TYPE_BPE: {
  GGML_ASSERT(false);
- return unicode_to_bytes_bpe(token_data.text);
+ return unicode_utf8_to_byte(token_data.text);
  }
  case LLAMA_VOCAB_TYPE_WPM: {
  GGML_ASSERT(false);
@@ -9375,7 +9375,7 @@ static llama_token llama_byte_to_token(const llama_vocab & vocab, uint8_t ch) {
  }
  case LLAMA_VOCAB_TYPE_WPM:
  case LLAMA_VOCAB_TYPE_BPE: {
- return vocab.token_to_id.at(bytes_to_unicode_bpe(ch));
+ return vocab.token_to_id.at(unicode_byte_to_utf8(ch));
  }
  default:
  GGML_ASSERT(false);
@@ -9715,9 +9715,9 @@ struct llm_tokenizer_bpe {
  bpe_words.reserve(text.size());
  bpe_encoded_words.reserve(text.size());
 
- auto cps = codepoints_from_utf8(text);
- for (size_t i = 0; i < cps.size(); ++i)
- text_utf.emplace_back(codepoint_to_utf8(cps[i]));
+ const auto cpts = unicode_cpts_from_utf8(text);
+ for (size_t i = 0; i < cpts.size(); ++i)
+ text_utf.emplace_back(unicode_cpt_to_utf8(cpts[i]));
 
  for (int i = 0; i < (int)text_utf.size(); i++) {
  const std::string & utf_char = text_utf[i];
@@ -9767,40 +9767,40 @@ struct llm_tokenizer_bpe {
  }
 
  if (!split_condition && !collecting) {
- if (codepoint_type(utf_char) == CODEPOINT_TYPE_LETTER || (!token.size() && utf_char == " " && codepoint_type(utf_char_next) == CODEPOINT_TYPE_LETTER)) {
+ if (unicode_cpt_type(utf_char) == CODEPOINT_TYPE_LETTER || (!token.size() && utf_char == " " && unicode_cpt_type(utf_char_next) == CODEPOINT_TYPE_LETTER)) {
  collecting_letter = true;
  collecting = true;
  }
- else if (codepoint_type(utf_char) == CODEPOINT_TYPE_DIGIT || (!token.size() && utf_char == " " && codepoint_type(utf_char_next) == CODEPOINT_TYPE_DIGIT)) {
+ else if (unicode_cpt_type(utf_char) == CODEPOINT_TYPE_DIGIT || (!token.size() && utf_char == " " && unicode_cpt_type(utf_char_next) == CODEPOINT_TYPE_DIGIT)) {
  collecting_numeric = true;
  collecting = true;
  }
  else if (
- ((codepoint_type(utf_char) != CODEPOINT_TYPE_LETTER && codepoint_type(utf_char) != CODEPOINT_TYPE_DIGIT) && (codepoint_type(utf_char) != CODEPOINT_TYPE_WHITESPACE)) ||
- (!token.size() && utf_char == " " && codepoint_type(utf_char_next) != CODEPOINT_TYPE_LETTER && codepoint_type(utf_char_next) != CODEPOINT_TYPE_DIGIT && codepoint_type(utf_char_next) != CODEPOINT_TYPE_WHITESPACE)
+ ((unicode_cpt_type(utf_char) != CODEPOINT_TYPE_LETTER && unicode_cpt_type(utf_char) != CODEPOINT_TYPE_DIGIT) && (unicode_cpt_type(utf_char) != CODEPOINT_TYPE_WHITESPACE)) ||
+ (!token.size() && utf_char == " " && unicode_cpt_type(utf_char_next) != CODEPOINT_TYPE_LETTER && unicode_cpt_type(utf_char_next) != CODEPOINT_TYPE_DIGIT && unicode_cpt_type(utf_char_next) != CODEPOINT_TYPE_WHITESPACE)
  ) {
  collecting_special = true;
  collecting = true;
  }
- else if (codepoint_type(utf_char) == CODEPOINT_TYPE_WHITESPACE && codepoint_type(utf_char_next) == CODEPOINT_TYPE_WHITESPACE) {
+ else if (unicode_cpt_type(utf_char) == CODEPOINT_TYPE_WHITESPACE && unicode_cpt_type(utf_char_next) == CODEPOINT_TYPE_WHITESPACE) {
  collecting_whitespace_lookahead = true;
  collecting = true;
  }
- else if (codepoint_type(utf_char) == CODEPOINT_TYPE_WHITESPACE) {
+ else if (unicode_cpt_type(utf_char) == CODEPOINT_TYPE_WHITESPACE) {
  split_condition = true;
  }
  }
  else if (!split_condition && collecting) {
- if (collecting_letter && codepoint_type(utf_char) != CODEPOINT_TYPE_LETTER) {
+ if (collecting_letter && unicode_cpt_type(utf_char) != CODEPOINT_TYPE_LETTER) {
  split_condition = true;
  }
- else if (collecting_numeric && codepoint_type(utf_char) != CODEPOINT_TYPE_DIGIT) {
+ else if (collecting_numeric && unicode_cpt_type(utf_char) != CODEPOINT_TYPE_DIGIT) {
  split_condition = true;
  }
- else if (collecting_special && (codepoint_type(utf_char) == CODEPOINT_TYPE_LETTER || codepoint_type(utf_char) == CODEPOINT_TYPE_DIGIT || codepoint_type(utf_char) == CODEPOINT_TYPE_WHITESPACE)) {
+ else if (collecting_special && (unicode_cpt_type(utf_char) == CODEPOINT_TYPE_LETTER || unicode_cpt_type(utf_char) == CODEPOINT_TYPE_DIGIT || unicode_cpt_type(utf_char) == CODEPOINT_TYPE_WHITESPACE)) {
  split_condition = true;
  }
- else if (collecting_whitespace_lookahead && (codepoint_type(utf_char_next) == CODEPOINT_TYPE_LETTER || codepoint_type(utf_char_next) == CODEPOINT_TYPE_DIGIT)) {
+ else if (collecting_whitespace_lookahead && (unicode_cpt_type(utf_char_next) == CODEPOINT_TYPE_LETTER || unicode_cpt_type(utf_char_next) == CODEPOINT_TYPE_DIGIT)) {
  split_condition = true;
  }
  }
@@ -9829,7 +9829,7 @@ struct llm_tokenizer_bpe {
  for (std::string & word : bpe_words) {
  std::string encoded_token = "";
  for (char & c : word) {
- encoded_token += bytes_to_unicode_bpe(c);
+ encoded_token += unicode_byte_to_utf8(c);
  }
  bpe_encoded_words.emplace_back(encoded_token);
  }
@@ -9903,33 +9903,21 @@ struct llm_tokenizer_wpm {
  }
 
  std::vector<std::string> preprocess(const std::string & text) {
- // normalalization form D
- std::vector<uint32_t> codepoints = codepoints_from_utf8(text);
- std::vector<uint32_t> nfd_codepoints;
- for (uint32_t code : codepoints) {
- auto it = nfd_map.equal_range(code);
- if (it.first != it.second) {
- for (auto jt = it.first; jt != it.second; jt++) {
- nfd_codepoints.push_back(jt->second);
- }
- } else {
- nfd_codepoints.push_back(code);
- }
- }
+ std::vector<uint32_t> cpts_nfd = unicode_cpts_normalize_nfd(unicode_cpts_from_utf8(text));
 
  // strip accents, strip control, uniformize whitespace,
  // to lowercase, pad chinese characters, pad punctuation
  std::string new_str = "";
- for (uint32_t code : nfd_codepoints) {
- int type = codepoint_type(code);
+ for (uint32_t code : cpts_nfd) {
+ int type = unicode_cpt_type(code);
  if (type == CODEPOINT_TYPE_ACCENT_MARK || type == CODEPOINT_TYPE_CONTROL) {
  continue;
  }
  code = to_lower(code);
  if (type == CODEPOINT_TYPE_WHITESPACE) {
  code = ' ';
  }
- std::string s = codepoint_to_utf8(code);
+ std::string s = unicode_cpt_to_utf8(code);
  if (type == CODEPOINT_TYPE_PUNCTUATION || is_ascii_punct(code) || is_chinese_char(code)) {
  new_str += " ";
  new_str += s;
@@ -9949,8 +9937,7 @@ struct llm_tokenizer_wpm {
  if (r > l) words.push_back(new_str.substr(l, (r - l)));
  l = r + 1;
  r = l;
- }
- else {
+ } else {
  r += 1;
  }
  }
@@ -9974,17 +9961,17 @@ struct llm_tokenizer_wpm {
  return code < 256 && ispunct(code);
  }
 
- bool is_chinese_char(uint32_t codepoint) {
- if ((codepoint >= 0x4E00 && codepoint <= 0x9FFF) ||
- (codepoint >= 0x3400 && codepoint <= 0x4DBF) ||
- (codepoint >= 0x20000 && codepoint <= 0x2A6DF) ||
- (codepoint >= 0x2A700 && codepoint <= 0x2B73F) ||
- (codepoint >= 0x2B740 && codepoint <= 0x2B81F) ||
- (codepoint >= 0x2B920 && codepoint <= 0x2CEAF) || // this should be 0x2B820 but in hf rust code it is 0x2B920
- (codepoint >= 0xF900 && codepoint <= 0xFAFF) ||
- (codepoint >= 0x2F800 && codepoint <= 0x2FA1F) ||
- (codepoint >= 0x3000 && codepoint <= 0x303F) ||
- (codepoint >= 0xFF00 && codepoint <= 0xFFEF)) {
+ bool is_chinese_char(uint32_t cpt) {
+ if ((cpt >= 0x4E00 && cpt <= 0x9FFF) ||
+ (cpt >= 0x3400 && cpt <= 0x4DBF) ||
+ (cpt >= 0x20000 && cpt <= 0x2A6DF) ||
+ (cpt >= 0x2A700 && cpt <= 0x2B73F) ||
+ (cpt >= 0x2B740 && cpt <= 0x2B81F) ||
+ (cpt >= 0x2B920 && cpt <= 0x2CEAF) || // this should be 0x2B820 but in hf rust code it is 0x2B920
+ (cpt >= 0xF900 && cpt <= 0xFAFF) ||
+ (cpt >= 0x2F800 && cpt <= 0x2FA1F) ||
+ (cpt >= 0x3000 && cpt <= 0x303F) ||
+ (cpt >= 0xFF00 && cpt <= 0xFFEF)) {
  return true; // NOLINT
  }
  return false;
@@ -13966,9 +13953,9 @@ int32_t llama_tokenize(
 
 static std::string llama_decode_text(const std::string & text) {
  std::string decoded_text;
- auto unicode_sequences = codepoints_from_utf8(text);
- for (auto& unicode_sequence : unicode_sequences) {
- decoded_text += unicode_to_bytes_bpe(codepoint_to_utf8(unicode_sequence));
+ auto unicode_sequences = unicode_cpts_from_utf8(text);
+ for (auto & unicode_sequence : unicode_sequences) {
+ decoded_text += unicode_utf8_to_byte(unicode_cpt_to_utf8(unicode_sequence));
  }
 
  return decoded_text;

diff --git a/tests/test-tokenizer-1-bpe.cpp b/tests/test-tokenizer-1-bpe.cpp
@@ -64,7 +64,7 @@ int main(int argc, char **argv) {
  for (int i = 0; i < n_vocab; ++i) {
  std::string str = llama_detokenize_bpe(ctx, std::vector<int>(1, i));
  try {
- auto cps = codepoints_from_utf8(str);
+ auto cps = unicode_cpts_from_utf8(str);
  std::vector<llama_token> tokens = llama_tokenize(ctx, str, false);
  std::string check = llama_detokenize_bpe(ctx, tokens);
  if (check != str) {
@@ -97,7 +97,7 @@ int main(int argc, char **argv) {
  continue;
  }
 
- std::string str = codepoint_to_utf8(cp);
+ std::string str = unicode_cpt_to_utf8(cp);
  std::vector<llama_token> tokens = llama_tokenize(ctx, str, false);
  std::string check = llama_detokenize_bpe(ctx, tokens);
  if (cp != 9601 && str != check) {

diff --git a/tests/test-tokenizer-1-llama.cpp b/tests/test-tokenizer-1-llama.cpp
@@ -85,7 +85,7 @@ int main(int argc, char **argv) {
  continue;
  }
 
- std::string str = codepoint_to_utf8(cp);
+ std::string str = unicode_cpt_to_utf8(cp);
  std::vector<llama_token> tokens = llama_tokenize(ctx, str, false);
  std::string check = llama_detokenize_spm(ctx, tokens);
  if (cp != 9601 && str != check) {