Files
ollama/x/create/laguna.go
Daniel Hiltgen 87288ced4f New models (#15861)
* mlx: add laguna model support

* convert: support fp8 safetensors import

Decode HF F8_E4M3 safetensors with block scale companions into GGUF-supported tensor types, and record which output tensors came from FP8 source weights.

Use that source-precision metadata during create quantization: default FP8-sourced GGUFs to Q8_0, keep non-FP8 tensors at their original precision for Q8_0, and promote non-FP8 quantizable tensors to Q8_0 for Q4_K requests.

* ggml: add laguna model support

* server: preserve generate logprobs with builtin parsers

Generate requests were dropping logprob-only chunks whenever a builtin parser buffered visible content. Chat already handled this case, but generate only forwarded chunks with visible response, thinking, or tool-call output.

Keep generate chunks that carry logprobs even when the builtin parser has not flushed visible content yet, and add a regression test that exercises the behavior with a generic thinking parser.

* review comments - perf improvements

* ggml: implement nemotron 3 nano omni

* add poolside integration

* update poolside doc

* adapt to new cache setup

* fix test

* fix test

---------

Co-authored-by: Eva Ho <hoyyeva@gmail.com>
2026-04-28 11:50:12 -07:00

60 lines
1.5 KiB
Go

package create
import (
"strings"
"github.com/ollama/ollama/x/safetensors"
)
type lagunaImportTransform struct{}
func newLagunaImportTransform(string, sourceModelConfig) (tensorImportTransform, error) {
return lagunaImportTransform{}, nil
}
func (lagunaImportTransform) skipTensor(string) bool { return false }
func (lagunaImportTransform) transformTensor(td *safetensors.TensorData) ([]*safetensors.TensorData, error) {
if td == nil {
return nil, nil
}
return []*safetensors.TensorData{td}, nil
}
func (lagunaImportTransform) quantizationType(name string, shape []int32, quantize string) string {
if !lagunaIsHFRoutedExpertWeight(name) {
return ""
}
return GetTensorQuantization(name, shape, quantize)
}
func (lagunaImportTransform) sourceFP8TensorQuantization(name string, shape []int32, requested string, fallback string) string {
if !lagunaIsHFRoutedExpertWeight(name) {
return ""
}
switch normalizeQuantType(requested) {
case "nvfp4", "mxfp4":
if lagunaKeepSourceFP8TensorAtMXFP8(name, shape) {
return "mxfp8"
}
}
return fallback
}
func (lagunaImportTransform) sourceFP8BF16Quantization(string, []int32, string) string {
return ""
}
func lagunaKeepSourceFP8TensorAtMXFP8(name string, shape []int32) bool {
if len(shape) != 2 || !isAligned(shape, "mxfp8") {
return false
}
return strings.Contains(name, "down_proj")
}
func lagunaIsHFRoutedExpertWeight(name string) bool {
return strings.HasSuffix(name, ".weight") && strings.Contains(name, ".mlp.experts.")
}