update tokenizer

9194874d · Michael Yang · 9d1de41b · 9194874d · 9194874d
Commit 9194874d authored Jul 26, 2025 by Michael Yang
Show whitespace changes
Inline Side-by-side

Showing with 13 additions and 2 deletions

model/bytepairencoding.go model/bytepairencoding.go +1 -1

model/models/gptoss/model.go model/models/gptoss/model.go +12 -1

No files found.
--- a/model/bytepairencoding.go
+++ b/model/bytepairencoding.go
@@ -22,7 +22,7 @@ var _ TextProcessor = (*BytePairEncoding)(nil)

 func NewBytePairEncoding(pre string, vocab *Vocabulary) BytePairEncoding {
 	return BytePairEncoding{
-		pre:   regexp2.MustCompile(pre, regexp2.Unicode|regexp2.RE2),
+		pre:   regexp2.MustCompile(pre, regexp2.None),
 		vocab: vocab,
 	}
 }

--- a/model/models/gptoss/model.go
+++ b/model/models/gptoss/model.go
@@ -3,6 +3,7 @@ package gptoss
 import (
 	"cmp"
 	"math"
+	"strings"

 	"github.com/ollama/ollama/fs"
 	"github.com/ollama/ollama/kvcache"
@@ -216,7 +217,17 @@ func New(c fs.Config) (model.Model, error) {
 	m := Transformer{
 		TransformerBlocks: make([]TransformerBlock, c.Uint("block_count")),
 		BytePairEncoding: model.NewBytePairEncoding(
-			c.String("tokenizer.ggml.pretokenizer", `(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\r\n\p{L}\p{N}]?\p{L}+|\p{N}{1,3}| ?[^\s\p{L}\p{N}]+[\r\n]*|\s*[\r\n]+|\s+(?!\S)|\s+`),
+			c.String("tokenizer.ggml.pretokenizer",
+				strings.Join([]string{
+					`[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]*[\p{Ll}\p{Lm}\p{Lo}\p{M}]+(?i:'s|'t|'re|'ve|'m|'ll|'d)?`,
+					`[^\r\n\p{L}\p{N}]?[\p{Lu}\p{Lt}\p{Lm}\p{Lo}\p{M}]+[\p{Ll}\p{Lm}\p{Lo}\p{M}]*(?i:'s|'t|'re|'ve|'m|'ll|'d)?`,
+					`\p{N}{1,3}`,
+					` ?[^\s\p{L}\p{N}]+[\r\n/]*`,
+					`\s*[\r\n]+`,
+					`\s+(?!\S)`,
+					`\s+`,
+				}, "|"),
+			),
 			&model.Vocabulary{
 				Values: c.Strings("tokenizer.ggml.tokens"),
 				Types:  c.Ints("tokenizer.ggml.token_type"),