Skip to content
3 changes: 3 additions & 0 deletions core/backend/options.go
Original file line number Diff line number Diff line change
Expand Up @@ -575,6 +575,9 @@ func gRPCPredictOpts(c config.ModelConfig, modelPath string) *pb.PredictOptions
metadata["enable_thinking"] = "true"
}
}
if c.ResponseFormat != "" {
metadata["response_format"] = c.ResponseFormat
}
// Forward the effective reasoning effort so the backend can pass it to the
// jinja chat template (chat_template_kwargs.reasoning_effort) — the lever
// models like gpt-oss / LFM2.5 actually read, distinct from enable_thinking.
Expand Down
12 changes: 12 additions & 0 deletions core/http/endpoints/openai/chat.go
Original file line number Diff line number Diff line change
Expand Up @@ -287,7 +287,9 @@ func ChatEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, evaluator
switch d.Type {
case "json_object":
input.Grammar = functions.JSONBNF
config.ResponseFormat = "json_object"
case "json_schema":
config.ResponseFormat = "json_schema"
d := schema.JsonSchemaRequest{}
dat, err := json.Marshal(config.ResponseFormatMap)
if err != nil {
Expand All @@ -297,6 +299,16 @@ func ChatEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, evaluator
if err != nil {
return err
}

// Pass raw JSON schema via metadata for backends that support native structured output
schemaBytes, err := json.Marshal(d.JsonSchema.Schema)
if err == nil {
if config.RequestMetadata == nil {
config.RequestMetadata = map[string]string{}
}
config.RequestMetadata["json_schema"] = string(schemaBytes)
}

fs := &functions.JSONFunctionStructure{
AnyOf: []functions.Item{d.JsonSchema.Schema},
}
Expand Down
28 changes: 27 additions & 1 deletion core/http/endpoints/openai/completion.go
Original file line number Diff line number Diff line change
Expand Up @@ -103,8 +103,34 @@ func CompletionEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, eva
d := schema.ChatCompletionResponseFormat{}
dat, _ := json.Marshal(config.ResponseFormatMap)
_ = json.Unmarshal(dat, &d)
if d.Type == "json_object" {
switch d.Type {
Comment thread
fury0928 marked this conversation as resolved.
case "json_object":
input.Grammar = functions.JSONBNF
config.ResponseFormat = "json_object"
case "json_schema":
config.ResponseFormat = "json_schema"
jsr := schema.JsonSchemaRequest{}
dat, err := json.Marshal(config.ResponseFormatMap)
if err == nil {
if err := json.Unmarshal(dat, &jsr); err == nil {
schemaBytes, err := json.Marshal(jsr.JsonSchema.Schema)
if err == nil {
if config.RequestMetadata == nil {
config.RequestMetadata = map[string]string{}
}
config.RequestMetadata["json_schema"] = string(schemaBytes)
}
fs := &functions.JSONFunctionStructure{
AnyOf: []functions.Item{jsr.JsonSchema.Schema},
}
g, err := fs.Grammar(config.FunctionsConfig.GrammarOptions()...)
if err == nil {
input.Grammar = g
} else {
xlog.Error("Failed generating grammar", "error", err)
}
}
}
}
}

Expand Down
37 changes: 35 additions & 2 deletions core/http/endpoints/openresponses/responses.go
Original file line number Diff line number Diff line change
Expand Up @@ -175,9 +175,42 @@ func ResponsesEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, eval
Functions: funcs,
}

// Handle text_format -> response_format conversion
// Handle text_format -> response_format conversion and structured output
if input.TextFormat != nil {
openAIReq.ResponseFormat = convertTextFormatToResponseFormat(input.TextFormat)
responseFormat := convertTextFormatToResponseFormat(input.TextFormat)
openAIReq.ResponseFormat = responseFormat

// Generate grammar and pass schema for structured output (like OpenAI chat/completion)
if rfMap, ok := responseFormat.(map[string]interface{}); ok {
if rfType, _ := rfMap["type"].(string); rfType == "json_object" {
cfg.Grammar = functions.JSONBNF
cfg.ResponseFormat = "json_object"
} else if rfType == "json_schema" {
cfg.ResponseFormat = "json_schema"
d := schema.JsonSchemaRequest{}
dat, err := json.Marshal(rfMap)
if err == nil {
if err := json.Unmarshal(dat, &d); err == nil {
schemaBytes, err := json.Marshal(d.JsonSchema.Schema)
if err == nil {
if cfg.RequestMetadata == nil {
cfg.RequestMetadata = map[string]string{}
}
cfg.RequestMetadata["json_schema"] = string(schemaBytes)
}
fs := &functions.JSONFunctionStructure{
AnyOf: []functions.Item{d.JsonSchema.Schema},
}
g, err := fs.Grammar(cfg.FunctionsConfig.GrammarOptions()...)
if err == nil {
cfg.Grammar = g
} else {
xlog.Error("Open Responses - Failed generating grammar for json_schema", "error", err)
}
}
}
}
}
}

// Generate grammar for function calling (similar to OpenAI chat endpoint)
Expand Down
96 changes: 95 additions & 1 deletion docs/content/features/constrained_grammars.md
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,11 @@ url = "/features/constrained_grammars/"
The `chat` endpoint supports the `grammar` parameter, which allows users to specify a grammar in Backus-Naur Form (BNF). This feature enables the Large Language Model (LLM) to generate outputs adhering to a user-defined schema, such as `JSON`, `YAML`, or any other format that can be defined using BNF. For more details about BNF, see [Backus-Naur Form on Wikipedia](https://en.wikipedia.org/wiki/Backus%E2%80%93Naur_form).

{{% notice note %}}
**Compatibility Notice:** This feature is only supported by models that use the [llama.cpp](https://github.com/ggerganov/llama.cpp) backend. For a complete list of compatible models, refer to the [Model Compatibility]({{%relref "reference/compatibility-table" %}}) page. For technical details, see the related pull requests: [PR #1773](https://github.com/ggerganov/llama.cpp/pull/1773) and [PR #1887](https://github.com/ggerganov/llama.cpp/pull/1887).
**Compatibility Notice:** Grammar and structured output support is available for the following backends:
- **llama.cpp** — supports the `grammar` parameter (GBNF syntax) and `response_format` with `json_schema`/`json_object`
- **vLLM** — supports the `grammar` parameter (via xgrammar), `response_format` with `json_schema` (native JSON schema enforcement), and `json_object`

For a complete list of compatible models, refer to the [Model Compatibility]({{%relref "reference/compatibility-table" %}}) page.
{{% /notice %}}

## Setup
Expand Down Expand Up @@ -66,6 +70,96 @@ For more complex grammars, you can define multi-line BNF rules. The grammar pars
- Character classes (`[a-z]`)
- String literals (`"text"`)

## vLLM Backend

The vLLM backend supports structured output via three methods:

### JSON Schema (recommended)

Use the OpenAI-compatible `response_format` parameter with `json_schema` to enforce a specific JSON structure:

```bash
curl http://localhost:8080/v1/chat/completions -H "Content-Type: application/json" -d '{
"model": "my-vllm-model",
"messages": [{"role": "user", "content": "Generate a person object"}],
"response_format": {
"type": "json_schema",
"json_schema": {
"name": "person",
"schema": {
"type": "object",
"properties": {
"name": {"type": "string"},
"age": {"type": "integer"}
},
"required": ["name", "age"]
}
}
}
}'
```

### JSON Object

Force the model to output valid JSON (without a specific schema):

```bash
curl http://localhost:8080/v1/chat/completions -H "Content-Type: application/json" -d '{
"model": "my-vllm-model",
"messages": [{"role": "user", "content": "Generate a person as JSON"}],
"response_format": {"type": "json_object"}
}'
```

### Grammar

The `grammar` parameter also works with vLLM via xgrammar:

```bash
curl http://localhost:8080/v1/chat/completions -H "Content-Type: application/json" -d '{
"model": "my-vllm-model",
"messages": [{"role": "user", "content": "Do you like apples?"}],
"grammar": "root ::= (\"yes\" | \"no\")"
}'
```

## Open Responses API

The Open Responses API (`/v1/responses`) also supports structured output via the `text_format` parameter:

### JSON Schema

```bash
curl http://localhost:8080/v1/responses -H "Content-Type: application/json" -d '{
"model": "my-model",
"input": "Generate a person object",
"text_format": {
"type": "json_schema",
"json_schema": {
"name": "person",
"schema": {
"type": "object",
"properties": {
"name": {"type": "string"},
"age": {"type": "integer"}
},
"required": ["name", "age"]
}
}
}
}'
```

### JSON Object

```bash
curl http://localhost:8080/v1/responses -H "Content-Type: application/json" -d '{
"model": "my-model",
"input": "Generate a person as JSON",
"text_format": {"type": "json_object"}
}'
```

## Related Features

- [OpenAI Functions]({{%relref "features/openai-functions" %}}) - Function calling with structured outputs
Expand Down