pick best template by capabilities

This commit is contained in:
Daniel Hiltgen 2026-05-20 17:37:47 -07:00
parent da36fff9f7
commit 7152996aaa
8 changed files with 200 additions and 80 deletions

View file

@ -214,7 +214,7 @@ func LogLevel() slog.Level {
var (
// FlashAttention enables the experimental flash attention feature.
FlashAttention = BoolWithDefault("OLLAMA_FLASH_ATTENTION")
// GoTemplate enables legacy Modelfile TEMPLATE rendering when a model has one.
// GoTemplate enables Modelfile TEMPLATE rendering when a model has one.
GoTemplate = BoolWithDefault("OLLAMA_GO_TEMPLATE")
// DebugLogRequests logs inference requests to disk for replay/debugging.
DebugLogRequests = Bool("OLLAMA_DEBUG_LOG_REQUESTS")

View file

@ -1,9 +1,9 @@
// llama_server.go wraps the llama-server binary as a subprocess
//
// Ollama uses two chat paths with llama-server. Models with explicit Ollama
// renderers/parsers, Harmony handling, MLX, or an enabled legacy TEMPLATE layer
// renderers/parsers, Harmony handling, MLX, or an enabled Go TEMPLATE layer
// still render prompts in Go and call /completion. Other GGUF chat models use
// llama-server's native chat template handling through /v1/chat/completions.
// llama-server's chat_template handling through /v1/chat/completions.
//
// For structured output, JSON schemas are passed directly to llama-server via
// its json_schema field (avoiding the CGO SchemaToGrammar dependency). Raw BNF

View file

@ -1427,7 +1427,7 @@ func TestAppendJinjaArgs(t *testing.T) {
want []string
}{
{
name: "native llama-server template path leaves jinja enabled",
name: "llama-server chat_template path leaves jinja enabled",
want: []string{"base"},
},
{

View file

@ -61,21 +61,22 @@ type registryOptions struct {
}
type Model struct {
Name string `json:"name"`
Config model.ConfigV2
ShortName string
ModelPath string
DraftPath string
ParentModel string
HasChatTemplate bool
HasLegacyTemplate bool
AdapterPaths []string
ProjectorPaths []string
System string
License []string
Digest string
Options map[string]any
Messages []api.Message
Name string `json:"name"`
Config model.ConfigV2
ShortName string
ModelPath string
DraftPath string
ParentModel string
HasChatTemplate bool
HasGoTemplate bool
PreferChatTemplate bool // set when GGUF chat_template has more capabilities than Go TEMPLATE
AdapterPaths []string
ProjectorPaths []string
System string
License []string
Digest string
Options map[string]any
Messages []api.Message
Template *template.Template
}
@ -159,7 +160,7 @@ func chatTemplateCapabilities(capabilities []model.Capability, chatTemplate stri
return capabilities
}
if strings.Contains(chatTemplate, "tools") || strings.Contains(chatTemplate, "tool_call") {
if chatTemplateHasToolSupport(chatTemplate) {
capabilities = appendCapability(capabilities, model.CapabilityTools)
}
if strings.Contains(chatTemplate, "<think>") && strings.Contains(chatTemplate, "</think>") {
@ -169,6 +170,45 @@ func chatTemplateCapabilities(capabilities []model.Capability, chatTemplate stri
return capabilities
}
func chatTemplateHasToolSupport(chatTemplate string) bool {
return strings.Contains(chatTemplate, "tools") || strings.Contains(chatTemplate, "tool_call")
}
func goTemplateCapabilities(t *template.Template) []model.Capability {
if t == nil {
return nil
}
v, err := t.Vars()
if err != nil {
slog.Warn("model template contains errors", "error", err)
return nil
}
var capabilities []model.Capability
if slices.Contains(v, "tools") {
capabilities = appendCapability(capabilities, model.CapabilityTools)
}
if slices.Contains(v, "suffix") {
capabilities = appendCapability(capabilities, model.CapabilityInsert)
}
openingTag, closingTag := thinking.InferTags(t.Template)
if openingTag != "" && closingTag != "" {
capabilities = appendCapability(capabilities, model.CapabilityThinking)
}
return capabilities
}
func hasMoreCapabilities(candidate, current []model.Capability) bool {
return len(candidate) > len(current)
}
func goTemplateEnvSet() bool {
return envconfig.GoTemplate(true) == envconfig.GoTemplate(false)
}
func (m *Model) projectorCapabilities(capabilities []model.Capability) []model.Capability {
if len(m.ProjectorPaths) == 0 {
return capabilities
@ -191,24 +231,12 @@ func (m *Model) projectorCapabilities(capabilities []model.Capability) []model.C
}
func (m *Model) templateCapabilities(capabilities []model.Capability) []model.Capability {
if m.Template == nil {
if m.HasGoTemplate && !shouldUseGoTemplate(m) {
return capabilities
}
v, err := m.Template.Vars()
if err != nil {
slog.Warn("model template contains errors", "error", err)
}
if slices.Contains(v, "tools") {
capabilities = appendCapability(capabilities, model.CapabilityTools)
}
if slices.Contains(v, "suffix") {
capabilities = appendCapability(capabilities, model.CapabilityInsert)
}
openingTag, closingTag := thinking.InferTags(m.Template.Template)
if openingTag != "" && closingTag != "" {
capabilities = appendCapability(capabilities, model.CapabilityThinking)
for _, capability := range goTemplateCapabilities(m.Template) {
capabilities = appendCapability(capabilities, capability)
}
return capabilities
@ -462,6 +490,7 @@ func GetModel(name string) (*Model, error) {
}
modelHasPooling := false
ggufChatTemplate := ""
for _, layer := range mf.Layers {
filename, err := manifest.BlobsPath(layer.Digest)
if err != nil {
@ -478,7 +507,8 @@ func GetModel(name string) (*Model, error) {
slog.Error("couldn't open model file", "error", err)
break
}
m.HasChatTemplate = f.KeyValue("tokenizer.chat_template").String() != ""
ggufChatTemplate = f.KeyValue("tokenizer.chat_template").String()
m.HasChatTemplate = ggufChatTemplate != ""
modelHasPooling = f.KeyValue("pooling_type").Valid()
f.Close()
}
@ -494,7 +524,7 @@ func GetModel(name string) (*Model, error) {
m.ProjectorPaths = append(m.ProjectorPaths, filename)
case "application/vnd.ollama.image.prompt",
"application/vnd.ollama.image.template":
m.HasLegacyTemplate = true
m.HasGoTemplate = true
bts, err := os.ReadFile(filename)
if err != nil {
return nil, err
@ -541,7 +571,14 @@ func GetModel(name string) (*Model, error) {
}
}
if m.ModelPath != "" && m.isGGUF() && !modelHasPooling && !m.HasChatTemplate && (!m.HasLegacyTemplate || !envconfig.GoTemplate(true)) && m.Config.Renderer == "" && m.Config.Parser == "" && !shouldUseHarmony(m) {
ggufCaps := chatTemplateCapabilities(nil, ggufChatTemplate)
goCaps := goTemplateCapabilities(m.Template)
if !goTemplateEnvSet() && m.HasGoTemplate && ggufChatTemplate != "" && m.Config.Renderer == "" && m.Config.Parser == "" && !shouldUseHarmony(m) && hasMoreCapabilities(ggufCaps, goCaps) {
m.PreferChatTemplate = true
slog.Debug("using GGUF chat_template because it has stronger capabilities than Go TEMPLATE", "model", m.Name, "chat_template_capabilities", ggufCaps, "go_template_capabilities", goCaps)
}
if m.ModelPath != "" && m.isGGUF() && !modelHasPooling && !m.HasChatTemplate && (!m.HasGoTemplate || !envconfig.GoTemplate(true)) && m.Config.Renderer == "" && m.Config.Parser == "" && !shouldUseHarmony(m) {
slog.Warn("model is missing tokenizer.chat_template and Go TEMPLATE support is unavailable; chat responses may be poorly formatted", "model", m.Name, "env", "OLLAMA_GO_TEMPLATE=1")
}

View file

@ -60,7 +60,7 @@ func TestPruneLayersSkipsRecentOrphans(t *testing.T) {
func TestGetModelTemplateMetadata(t *testing.T) {
customTemplate := "CUSTOM {{ .Prompt }}"
t.Run("records chat template and legacy template layer", func(t *testing.T) {
t.Run("records chat template and Go TEMPLATE layer", func(t *testing.T) {
t.Setenv("OLLAMA_MODELS", t.TempDir())
t.Setenv("OLLAMA_GO_TEMPLATE", "")
@ -77,14 +77,80 @@ func TestGetModelTemplateMetadata(t *testing.T) {
if !m.HasChatTemplate {
t.Fatal("expected GGUF chat template to be detected")
}
if !m.HasLegacyTemplate {
t.Fatal("expected legacy template layer to be detected")
if !m.HasGoTemplate {
t.Fatal("expected Go TEMPLATE layer to be detected")
}
if got := m.Template.String(); got != customTemplate {
t.Fatalf("template = %q, want %q", got, customTemplate)
}
})
t.Run("prefers chat template when Go TEMPLATE has fewer capabilities", func(t *testing.T) {
t.Setenv("OLLAMA_MODELS", t.TempDir())
t.Setenv("OLLAMA_GO_TEMPLATE", "")
_, digest := createBinFile(t, ggml.KV{
"general.architecture": "llama",
"tokenizer.chat_template": "{% if tools %}{{ tools }}{% endif %}{{ messages[0]['content'] }}",
}, nil)
writeTestModelManifest(t, "chat-template-tools", digest, customTemplate)
m, err := GetModel("chat-template-tools")
if err != nil {
t.Fatal(err)
}
if !m.PreferChatTemplate {
t.Fatal("expected chat template to be preferred")
}
if got := m.CheckCapabilities(model.CapabilityTools); got != nil {
t.Fatalf("expected tools capability, got %v", got)
}
})
t.Run("respects explicit Go TEMPLATE enablement", func(t *testing.T) {
t.Setenv("OLLAMA_MODELS", t.TempDir())
t.Setenv("OLLAMA_GO_TEMPLATE", "1")
_, digest := createBinFile(t, ggml.KV{
"general.architecture": "llama",
"tokenizer.chat_template": "{% if tools %}{{ tools }}{% endif %}{{ messages[0]['content'] }}",
}, nil)
writeTestModelManifest(t, "go-template-forced", digest, customTemplate)
m, err := GetModel("go-template-forced")
if err != nil {
t.Fatal(err)
}
if m.PreferChatTemplate {
t.Fatal("expected explicit Go TEMPLATE setting to suppress chat_template preference")
}
if got := m.CheckCapabilities(model.CapabilityTools); got == nil {
t.Fatal("expected tools capability to be unavailable when Go TEMPLATE is explicitly enabled")
}
})
t.Run("respects explicit Go TEMPLATE disablement", func(t *testing.T) {
t.Setenv("OLLAMA_MODELS", t.TempDir())
t.Setenv("OLLAMA_GO_TEMPLATE", "0")
_, digest := createBinFile(t, ggml.KV{
"general.architecture": "llama",
"tokenizer.chat_template": "{% if tools %}{{ tools }}{% endif %}{{ messages[0]['content'] }}",
}, nil)
writeTestModelManifest(t, "go-template-disabled", digest, customTemplate)
m, err := GetModel("go-template-disabled")
if err != nil {
t.Fatal(err)
}
if m.PreferChatTemplate {
t.Fatal("expected explicit Go TEMPLATE setting to suppress chat_template preference")
}
if got := m.CheckCapabilities(model.CapabilityTools); got != nil {
t.Fatalf("expected tools capability from GGUF chat_template, got %v", got)
}
})
t.Run("records missing chat template", func(t *testing.T) {
t.Setenv("OLLAMA_MODELS", t.TempDir())
t.Setenv("OLLAMA_GO_TEMPLATE", "")
@ -101,8 +167,8 @@ func TestGetModelTemplateMetadata(t *testing.T) {
if m.HasChatTemplate {
t.Fatal("expected missing GGUF chat template")
}
if !m.HasLegacyTemplate {
t.Fatal("expected legacy template layer to be detected")
if !m.HasGoTemplate {
t.Fatal("expected Go TEMPLATE layer to be detected")
}
})
}
@ -139,7 +205,7 @@ func TestModelCapabilities(t *testing.T) {
"general.architecture": "llama",
}, []*ggml.Tensor{})
nativeToolTemplateModelPath, _ := createBinFile(t, ggml.KV{
ggufToolTemplateModelPath, _ := createBinFile(t, ggml.KV{
"general.architecture": "llama",
"tokenizer.chat_template": `{% if tools %}<tool_call>{{ tools }}</tool_call>{% endif %}<think>{{ messages[0]['content'] }}</think>`,
}, []*ggml.Tensor{})
@ -242,19 +308,19 @@ func TestModelCapabilities(t *testing.T) {
expectedCaps: []model.Capability{model.CapabilityCompletion, model.CapabilityTools},
},
{
name: "model with native chat template tools and thinking",
name: "model with GGUF chat_template tools and thinking",
model: Model{
ModelPath: nativeToolTemplateModelPath,
ModelPath: ggufToolTemplateModelPath,
},
expectedCaps: []model.Capability{model.CapabilityCompletion, model.CapabilityTools, model.CapabilityThinking},
},
{
name: "model with Go template ignores native chat template capabilities",
name: "model with Go TEMPLATE ignores GGUF chat_template capabilities",
model: Model{
ModelPath: nativeToolTemplateModelPath,
Template: chatTemplate,
HasLegacyTemplate: true,
HasChatTemplate: true,
ModelPath: ggufToolTemplateModelPath,
Template: chatTemplate,
HasGoTemplate: true,
HasChatTemplate: true,
},
expectedCaps: []model.Capability{model.CapabilityCompletion},
},
@ -367,7 +433,7 @@ func TestModelCapabilities(t *testing.T) {
},
},
{
name: "legacy gemma4 safetensors suppresses vision and audio",
name: "default gemma4 safetensors suppresses vision and audio",
model: Model{
Config: model.ConfigV2{
ModelFormat: "safetensors",

View file

@ -570,7 +570,7 @@ func (s *Server) GenerateHandler(c *gin.Context) {
}
leadingBOS = leadingBOSForModel(m)
} else {
// legacy flow
// Direct template execution flow.
if err := tmpl.Execute(&b, values); err != nil {
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
return
@ -2327,11 +2327,18 @@ func llamaServerConfigForModel(m *Model) llm.LlamaServerConfig {
}
func usesOllamaRenderedChat(m *Model) bool {
return m != nil && (m.Config.Renderer != "" || m.Config.Parser != "" || shouldUseHarmony(m) || shouldUseLegacyTemplate(m))
return m != nil && (m.Config.Renderer != "" || m.Config.Parser != "" || shouldUseHarmony(m) || shouldUseGoTemplate(m))
}
func shouldUseLegacyTemplate(m *Model) bool {
return m.HasLegacyTemplate && envconfig.GoTemplate(true)
func shouldUseGoTemplate(m *Model) bool {
if !m.HasGoTemplate {
return false
}
if goTemplateEnvSet() {
return envconfig.GoTemplate(true)
}
return !m.PreferChatTemplate && envconfig.GoTemplate(true)
}
func writeChatResponse(c *gin.Context, req api.ChatRequest, ch chan any) {

View file

@ -143,7 +143,7 @@ func TestGenerateWithBuiltinRenderer(t *testing.T) {
})
t.Run("custom template bypasses renderer", func(t *testing.T) {
// Test that providing a custom template uses the legacy flow
// Test that providing a custom template uses direct template execution.
w := createRequest(t, s.GenerateHandler, api.GenerateRequest{
Model: "test-renderer",
Prompt: "Write a hello world function",
@ -180,7 +180,7 @@ func TestGenerateWithBuiltinRenderer(t *testing.T) {
}
t.Run("suffix bypasses renderer", func(t *testing.T) {
// Test that providing a suffix uses the legacy flow
// Test that providing a suffix uses direct template execution.
w := createRequest(t, s.GenerateHandler, api.GenerateRequest{
Model: "test-suffix-renderer",
Prompt: "def add(",

View file

@ -194,20 +194,30 @@ func createMinimalGGUFModel(t *testing.T, s *Server, name string, kv ggml.KV, tm
func TestChatModeForModel(t *testing.T) {
t.Setenv("OLLAMA_GO_TEMPLATE", "")
if got := chatModeForModel(&Model{HasChatTemplate: true, HasLegacyTemplate: true}); got != chatExecutionModeRendered {
if got := chatModeForModel(&Model{HasChatTemplate: true, HasGoTemplate: true}); got != chatExecutionModeRendered {
t.Fatalf("chatModeForModel with default go template env = %v, want rendered", got)
}
t.Setenv("OLLAMA_GO_TEMPLATE", "0")
if got := chatModeForModel(&Model{HasChatTemplate: true, HasLegacyTemplate: true}); got != chatExecutionModeNative {
t.Fatalf("chatModeForModel with go template env disabled = %v, want native", got)
if got := chatModeForModel(&Model{HasChatTemplate: true, HasGoTemplate: true}); got != chatExecutionModeNative {
t.Fatalf("chatModeForModel with go template env disabled = %v, want chat_template route", got)
}
t.Setenv("OLLAMA_GO_TEMPLATE", "1")
if got := chatModeForModel(&Model{HasChatTemplate: true, HasLegacyTemplate: true}); got != chatExecutionModeRendered {
if got := chatModeForModel(&Model{HasChatTemplate: true, HasGoTemplate: true}); got != chatExecutionModeRendered {
t.Fatalf("chatModeForModel with go template env enabled = %v, want rendered", got)
}
t.Setenv("OLLAMA_GO_TEMPLATE", "1")
if got := chatModeForModel(&Model{HasChatTemplate: true, HasGoTemplate: true, PreferChatTemplate: true}); got != chatExecutionModeRendered {
t.Fatalf("chatModeForModel with explicit go template env and chat_template preference = %v, want rendered", got)
}
t.Setenv("OLLAMA_GO_TEMPLATE", "")
if got := chatModeForModel(&Model{HasChatTemplate: true, HasGoTemplate: true, PreferChatTemplate: true}); got != chatExecutionModeNative {
t.Fatalf("chatModeForModel with default go template env and chat_template preference = %v, want chat_template route", got)
}
t.Setenv("OLLAMA_GO_TEMPLATE", "0")
parserModel := &Model{Config: model.ConfigV2{Parser: "gemma4"}, HasChatTemplate: true}
if got := chatModeForModel(parserModel); got != chatExecutionModeRendered {
@ -239,22 +249,22 @@ func TestChatModeForModel(t *testing.T) {
t.Setenv("OLLAMA_GO_TEMPLATE", "")
if got := chatModeForModel(&Model{Config: model.ConfigV2{ModelFamily: "unknown"}, HasChatTemplate: true}); got != chatExecutionModeNative {
t.Fatalf("chatModeForModel without legacy template = %v, want native", got)
t.Fatalf("chatModeForModel without Go TEMPLATE = %v, want chat_template route", got)
}
if got := llamaServerConfigForModel(&Model{Config: model.ConfigV2{ModelFamily: "unknown"}, HasChatTemplate: true}); got.DisableJinja {
t.Fatalf("llamaServerConfigForModel with native chat template should not disable jinja")
t.Fatalf("llamaServerConfigForModel with GGUF chat_template should not disable jinja")
}
legacyModel := &Model{Config: model.ConfigV2{ModelFamily: "unknown"}, HasLegacyTemplate: true}
if got := chatModeForModel(legacyModel); got != chatExecutionModeRendered {
t.Fatalf("chatModeForModel with generic legacy template = %v, want rendered", got)
goTemplateModel := &Model{Config: model.ConfigV2{ModelFamily: "unknown"}, HasGoTemplate: true}
if got := chatModeForModel(goTemplateModel); got != chatExecutionModeRendered {
t.Fatalf("chatModeForModel with generic Go TEMPLATE = %v, want rendered", got)
}
if got := llamaServerConfigForModel(legacyModel); !got.DisableJinja {
if got := llamaServerConfigForModel(goTemplateModel); !got.DisableJinja {
t.Fatalf("llamaServerConfigForModel with Go TEMPLATE should disable jinja")
}
}
func TestChatHandlerNativeTemplateRoute(t *testing.T) {
func TestChatHandlerChatTemplateRoute(t *testing.T) {
t.Setenv("OLLAMA_CONTEXT_LENGTH", "4096")
t.Setenv("OLLAMA_GO_TEMPLATE", "")
gin.SetMode(gin.TestMode)
@ -262,7 +272,7 @@ func TestChatHandlerNativeTemplateRoute(t *testing.T) {
mock := mockRunner{
ChatFn: func(_ context.Context, req llm.ChatRequest, fn func(llm.ChatResponse)) error {
fn(llm.ChatResponse{
Message: api.Message{Role: "assistant", Content: "native response"},
Message: api.Message{Role: "assistant", Content: "chat template response"},
Done: true,
DoneReason: llm.DoneReasonStop,
PromptEvalCount: 1,
@ -274,13 +284,13 @@ func TestChatHandlerNativeTemplateRoute(t *testing.T) {
},
}
s := newServerWithMockRunner(t, &mock)
createMinimalGGUFModel(t, s, "native-chat", ggml.KV{
createMinimalGGUFModel(t, s, "chat-template", ggml.KV{
"tokenizer.chat_template": "{{ messages[0]['content'] }}",
}, "", nil)
stream := false
w := createRequest(t, s.ChatHandler, api.ChatRequest{
Model: "native-chat",
Model: "chat-template",
Messages: []api.Message{
{Role: "user", Content: "hello"},
},
@ -294,14 +304,14 @@ func TestChatHandlerNativeTemplateRoute(t *testing.T) {
if err := json.Unmarshal(w.Body.Bytes(), &actual); err != nil {
t.Fatal(err)
}
if actual.Message.Content != "native response" {
t.Fatalf("expected native response, got %q", actual.Message.Content)
if actual.Message.Content != "chat template response" {
t.Fatalf("expected chat template response, got %q", actual.Message.Content)
}
if len(mock.ChatRequest.Messages) != 1 || mock.ChatRequest.Messages[0].Content != "hello" {
t.Fatalf("native chat request messages = %#v", mock.ChatRequest.Messages)
t.Fatalf("chat_template request messages = %#v", mock.ChatRequest.Messages)
}
if !mock.ChatRequest.Shift {
t.Fatal("expected native chat to preserve default cache_prompt shift")
t.Fatal("expected chat_template route to preserve default cache_prompt shift")
}
}
@ -312,19 +322,19 @@ func TestChatHandlerTemplateEnvUsesRenderedRoute(t *testing.T) {
mock := mockRunner{
CompletionResponse: llm.CompletionResponse{
Content: "legacy response",
Content: "go template response",
Done: true,
DoneReason: llm.DoneReasonStop,
},
}
s := newServerWithMockRunner(t, &mock)
createMinimalGGUFModel(t, s, "legacy-template", ggml.KV{
createMinimalGGUFModel(t, s, "go-template", ggml.KV{
"tokenizer.chat_template": "{{ messages[0]['content'] }}",
}, "{{ range .Messages }}{{ .Role }}: {{ .Content }}\n{{ end }}", nil)
stream := false
w := createRequest(t, s.ChatHandler, api.ChatRequest{
Model: "legacy-template",
Model: "go-template",
Messages: []api.Message{
{Role: "user", Content: "hello"},
},
@ -334,7 +344,7 @@ func TestChatHandlerTemplateEnvUsesRenderedRoute(t *testing.T) {
t.Fatalf("expected status 200, got %d: %s", w.Code, w.Body.String())
}
if mock.ChatRequest.Messages != nil {
t.Fatalf("expected rendered route, native chat request was recorded: %#v", mock.ChatRequest)
t.Fatalf("expected rendered route, chat_template request was recorded: %#v", mock.ChatRequest)
}
if !strings.Contains(mock.CompletionRequest.Prompt, "user: hello") {
t.Fatalf("expected rendered prompt, got %q", mock.CompletionRequest.Prompt)