Merge pull request #9703 from ollama/mxyng/gemma3-memory

count gemma3 vision tensors

Merge pull request #9703 from ollama/mxyng/gemma3-memory
count gemma3 vision tensors
4ea4d2b1 · Michael Yang · GitHub · 74b44fdf · 8d76fa23 · 4ea4d2b1
Unverified Commit 4ea4d2b1 authored Mar 13, 2025 by Michael Yang Committed by GitHub Mar 13, 2025
Hide whitespace changes
Inline Side-by-side

Showing with 34 additions and 21 deletions

fs/ggml/ggml.go fs/ggml/ggml.go +32 -19

llm/memory.go llm/memory.go +2 -2

No files found.
--- a/fs/ggml/ggml.go
+++ b/fs/ggml/ggml.go
@@ -583,39 +583,52 @@ func (f GGML) GraphSize(context, batch uint64, kvCacheType string) (kv, partialO
 }
 func (llm GGML) VisionGraphSize() (weights, graphSize uint64) {
-	switch llm.KV().Architecture() {
+	if llm.KV().Uint("vision.block_count") == 0 {
-	case "mllama":
+		return
-		for _, layer := range llm.Tensors().GroupLayers()["v"] {
+	}
-			weights += layer.Size()
-		}
-		kv := func(n string) uint64 {
+	for name, layer := range llm.Tensors().GroupLayers() {
-			if v, ok := llm.KV()["mllama.vision."+n].(uint32); ok {
+		if name == "v" || strings.HasPrefix(name, "v.") {
-				return uint64(v)
+			for _, tensor := range layer {
+				weights += tensor.Size()
 			}
-			return 0
 		}
+	}
+	imageSize := uint64(llm.KV().Uint("vision.image_size"))
+	patchSize := uint64(llm.KV().Uint("vision.patch_size"))
+	if patchSize == 0 {
+		slog.Warn("unknown patch size for vision model")
+		return
+	}
-		imageSize := kv("image_size")
+	numChannels := uint64(llm.KV().Uint("vision.num_channels"))
-		maxNumTiles := kv("max_num_tiles")
+	numPatches := (imageSize / patchSize) * (imageSize / patchSize)
-		embeddingLength := kv("embedding_length")
+	if _, ok := llm.Tensors().GroupLayers()["v"]["class_embd"]; ok {
-		headCount := kv("attention.head_count")
+		numPatches++
+	}
-		numPatches := (imageSize / kv("patch_size")) * (imageSize / kv("patch_size"))
+	headCount := uint64(llm.KV().Uint("vision.attention.head_count"))
-		if _, ok := llm.Tensors().GroupLayers()["v"]["class_embd"]; ok {
+	embeddingLength := uint64(llm.KV().Uint("vision.embedding_length"))
-			numPatches++
-		}
+	switch llm.KV().Architecture() {
+	case "mllama":
 		numPaddedPatches := numPatches + 8 - (numPatches%8)%8
+		maxNumTiles := uint64(llm.KV().Uint("vision.max_num_tiles"))
 		graphSize = 4 * (8 +
-			imageSize*imageSize*kv("num_channels")*maxNumTiles +
+			imageSize*imageSize*numChannels*maxNumTiles +
 			embeddingLength*numPatches*maxNumTiles +
 			9*embeddingLength*numPaddedPatches*maxNumTiles +
 			numPaddedPatches*maxNumTiles*numPaddedPatches*maxNumTiles*headCount)
+	case "gemma3":
+		graphSize = 4 * (imageSize*imageSize*numChannels +
+			embeddingLength*patchSize +
+			numPatches*numPatches*headCount)
 	}
 	return weights, graphSize
 }

--- a/llm/memory.go
+++ b/llm/memory.go
@@ -218,8 +218,8 @@ func EstimateGPULayers(gpus []discover.GpuInfo, f *ggml.GGML, projectors []strin
 		if blk, ok := layers[fmt.Sprintf("blk.%d", i)]; ok {
 			layerSize = blk.Size()
 			layerSize += kv / f.KV().BlockCount()
+			memoryWeights += blk.Size()
 		}
-		memoryWeights += layerSize
 		if opts.NumGPU >= 0 && layerCount >= opts.NumGPU {
 			// Stop allocating on GPU(s) once we hit the users target NumGPU
@@ -376,7 +376,7 @@ func (m MemoryEstimate) LogValue() slog.Value {
 				// memory of the weights
 				"total", format.HumanBytes2(m.memoryWeights),
 				// memory of repeating layers
-				"repeating", format.HumanBytes2(m.memoryWeights-m.memoryLayerOutput),
+				"repeating", format.HumanBytes2(m.memoryWeights),
 				// memory of non-repeating layers
 				"nonrepeating", format.HumanBytes2(m.memoryLayerOutput),
 			),