diff --git a/CHANGELOG.md b/CHANGELOG.md index d5e7286..17d2502 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,17 @@ All notable changes to GopherAgent are documented here. Format follows [Keep a Changelog](https://keepachangelog.com/); versions follow [Semantic Versioning](https://semver.org/) — pre-1.0, breaking API changes only require a minor bump. +## [v0.42.0] — 2026-08-15 + +### Added + +- **`openai.JSONMode` — a compatible endpoint now states which JSON mode it implements instead of being assumed into OpenAI's.** "OpenAI-compatible" describes the endpoint, the request and response shape, and the SDK; it does not promise every extension built on top of them. Structured output is where that gap bites: `api.openai.com` takes `response_format {type:"json_schema"}` with the schema inline and enforces it server-side, while several compatible endpoints publish only the older `{type:"json_object"}`, which guarantees JSON syntax and carries no schema field at all. The adapter sent `json_schema` unconditionally, so on such an endpoint every schema-constrained call was a 400 while plain chat and tool calling kept working — the failure landed exactly on the planner, extractor, and judge stages that depend on structured output, and nowhere else. `WithJSONMode` takes `JSONModeSchema`, `JSONModeObject`, or `JSONModeNone`; `WithImageInput` declares the other feature a gateway may or may not have. Under `JSONModeObject` the schema moves into the prompt as a trailing system message, appended last rather than merged into an existing system prompt because the schema is an instruction and the final message is the one models follow most closely. The rendered text always contains the literal word JSON, which is load-bearing rather than stylistic: endpoints implementing this format commonly reject a request whose messages never mention it, to stop callers switching on JSON mode and then asking for prose. What `JSONModeObject` cannot do is enforce anything — the endpoint guarantees only that the reply parses, so schema conformance degrades from a server-side constraint to a model instruction and `Strict` becomes a request rather than a rule. Callers on that path must validate the result and be ready to retry; the adapter does not retry on their behalf, because how many attempts a malformed reply is worth is policy the caller owns. (`pkg/llm/openai/json_mode.go`) + +### Changed (breaking) + +- **`openai.NewCompat` no longer claims image input or structured output on the endpoint's behalf.** `Capabilities()` was a method on the shared `*Provider` type returning a hardcoded `{ImageInput: true, StructuredOutput: true}`, and `NewCompat` returns that same type — so a provider pointed at any gateway in the ecosystem reported full support regardless of what the gateway actually implemented. That inverts the entire point of the signal. `CapabilityProvider` exists so a consumer can reject an unsuitable provider at construction instead of discovering the gap from a confident, wrong answer, and for compatible endpoints it produced precisely that: a pre-flight check that passed, followed by a failure mid-run. The report now derives from configuration — `StructuredOutput` is true whenever a JSON mode is declared — so a single source of truth governs both what is claimed and what goes on the wire, and the two cannot drift. `New` is unchanged and still reports both, since it does speak for `api.openai.com`. `NewCompat` starts from `JSONModeNone` and no image claim, in the same spirit as its already requiring an explicit model: a compatible endpoint has no sensible default, and the adapter declares nothing it cannot know. Callers using structured output through `NewCompat` must add `WithJSONMode(...)`; the alternative was to keep defaulting to `json_schema`, which preserves both the false claim and the 400. +- **A structured-output call against `JSONModeNone` fails before the request is sent.** The error names the option that fixes it. Silently dropping the constraint was the worse option in both directions: the model returns prose, and the caller's unmarshal fails somewhere unrelated to the cause — while forwarding a `response_format` the endpoint never published produces a 400 that names neither. (`pkg/llm/openai/openai.go`, `pkg/llm/openai/openai_compat.go`) + ## [v0.41.0] — 2026-08-10 ### Added @@ -635,6 +646,7 @@ Multi-user, long-running, audit-friendly chat surface — the foundation for sid - README section on the permission flow — documents `RequiresConfirmation` × `ConfirmHITL` × `Permissions` interaction. - Enum struct tag support in `tools.SchemaFor[T]()` — emit values into JSON-Schema's `enum` array so providers reject invalid values upstream. +[v0.42.0]: https://github.com/hung12ct/gopheragent/releases/tag/v0.42.0 [v0.41.0]: https://github.com/hung12ct/gopheragent/releases/tag/v0.41.0 [v0.40.0]: https://github.com/hung12ct/gopheragent/releases/tag/v0.40.0 [v0.39.0]: https://github.com/hung12ct/gopheragent/releases/tag/v0.39.0 diff --git a/docs/providers.md b/docs/providers.md index f193261..48e42ba 100644 --- a/docs/providers.md +++ b/docs/providers.md @@ -27,6 +27,43 @@ Compatible base URLs must be absolute HTTP(S) URLs without embedded credentials, query parameters, or fragments. Use HTTPS for remote gateways; plain HTTP remains available for local Ollama/vLLM development. +### Declare what the endpoint implements + +"OpenAI-compatible" covers the endpoint, the request/response shape, and the +SDK — not every extension built on top of them. `NewCompat` therefore claims +nothing beyond plain chat and tool calling, in the same spirit as requiring an +explicit model: the adapter cannot infer the rest from a base URL. + +Two declarations matter: + +| Option | Effect | +|---|---| +| `WithJSONMode(JSONModeSchema)` | `response_format {type:"json_schema"}` with the schema inline, enforced server-side. What `api.openai.com` implements. | +| `WithJSONMode(JSONModeObject)` | `response_format {type:"json_object"}` plus the schema appended to the prompt, for endpoints publishing only the older format. | +| `WithImageInput(true)` | Declares that the endpoint accepts image parts on user messages. | + +```go +// Endpoint publishing OpenAI's json_schema: +p, _ := openai.NewCompat(key, "openai/gpt-4o", "https://openrouter.ai/api/v1", + openai.WithJSONMode(openai.JSONModeSchema), openai.WithImageInput(true)) + +// Endpoint publishing only json_object: +p, _ := openai.NewCompat(key, "deepseek-v4-flash", "https://api.deepseek.com/v1", + openai.WithJSONMode(openai.JSONModeObject)) +``` + +Without a declaration the JSON mode is `JSONModeNone`, and a structured-output +call fails with an error naming the option instead of sending a +`response_format` the endpoint may reject. Check the endpoint's own docs for +which formats it publishes — the two are not interchangeable, and sending the +wrong one is a 400 on every structured call. + +`JSONModeObject` guarantees only that the reply parses as JSON. Schema +conformance degrades from a server-side constraint to a prompt instruction, so +validate the result and be ready to retry; `Strict` becomes a request rather +than a rule. This is the same trade the Anthropic adapter avoids by +synthesizing a tool — an option `json_object` endpoints do not offer. + ### Point the non-chat clients at the same endpoint `NewCompat` configures the **chat provider only**. The embedder, vision @@ -52,9 +89,10 @@ silently does nothing. `agent.CapabilityProvider` lets a consumer that requires image input or structured output reject an unsuitable provider at construction, instead of -discovering the gap from a confident, wrong answer. The OpenAI, Anthropic, and -Gemini adapters report both; `llmfake.ScriptedProvider` reports neither, since -it replays a script without ever reading a message's media parts. +discovering the gap from a confident, wrong answer. `openai.New`, Anthropic, +and Gemini report both; `openai.NewCompat` reports what the caller declared +(see above); `llmfake.ScriptedProvider` reports neither, since it replays a +script without ever reading a message's media parts. ```go if c, ok := provider.(agent.CapabilityProvider); ok && !c.Capabilities().ImageInput { diff --git a/pkg/agent/structured_output.go b/pkg/agent/structured_output.go index 1a99acb..ca3449a 100644 --- a/pkg/agent/structured_output.go +++ b/pkg/agent/structured_output.go @@ -10,6 +10,10 @@ import ( // // - OpenAI (gpt-4o-*, o-series): response_format: {type:"json_schema", // json_schema:{name, description, schema, strict}}. Native and strict. +// Compatible endpoints publishing only the older {type:"json_object"} +// declare it with openai.WithJSONMode, which moves the schema into the +// prompt: still JSON, but conformance stops being enforced server-side, +// so validate the result. // - Gemini (1.5+): response_mime_type:"application/json" + // response_schema. Native; Strict is ignored (Gemini always enforces). // - Anthropic Claude: no native JSON mode. Providers synthesize a single diff --git a/pkg/llm/openai/json_mode.go b/pkg/llm/openai/json_mode.go new file mode 100644 index 0000000..e251b6a --- /dev/null +++ b/pkg/llm/openai/json_mode.go @@ -0,0 +1,154 @@ +package openai + +import ( + "context" + "fmt" + "strings" + + "github.com/hung12ct/gopheragent/pkg/agent" + "github.com/sashabaranov/go-openai" +) + +// JSONMode selects the wire encoding an endpoint accepts for a +// schema-constrained response. +// +// api.openai.com takes response_format {type:"json_schema"} with the schema +// inline and enforces it server-side. Compatible endpoints implement an +// unknown subset of that: several publish only the older +// {type:"json_object"}, which guarantees JSON syntax but carries no schema +// field at all, and some publish neither. Sending json_schema to an endpoint +// that does not implement it is a 400 on every structured call, so the mode +// is configuration — the adapter cannot infer it from a base URL. +type JSONMode int + +const ( + // JSONModeSchema sends response_format {type:"json_schema", ...} with the + // schema inline, and the endpoint enforces it. Zero value because it is + // what this package is named for; NewCompat overrides it. + JSONModeSchema JSONMode = iota + // JSONModeObject sends response_format {type:"json_object"} and appends + // the schema to the prompt, because that wire format has nowhere to put + // it. The endpoint guarantees only that the reply parses as JSON — + // conformance to the schema degrades to a model instruction, so callers + // should validate the result and be ready to retry. + JSONModeObject + // JSONModeNone declares no JSON mode. A structured-output request against + // such an endpoint fails at the call, rather than silently downgrading to + // free-form prose that the caller would then try to unmarshal. + JSONModeNone +) + +// String implements fmt.Stringer so the mode reads as its constant name in +// error messages instead of an integer. +func (m JSONMode) String() string { + switch m { + case JSONModeSchema: + return "JSONModeSchema" + case JSONModeObject: + return "JSONModeObject" + case JSONModeNone: + return "JSONModeNone" + default: + return fmt.Sprintf("JSONMode(%d)", int(m)) + } +} + +// WithJSONMode declares which JSON mode this endpoint implements. It is +// meaningful only on a compatible endpoint: New already reports the mode +// api.openai.com implements, and NewCompat claims none until this says +// otherwise. +// +// It also drives Capabilities().StructuredOutput, so declaring the mode is +// what lets a consumer that depends on schema enforcement accept the +// provider at construction. +func WithJSONMode(m JSONMode) Option { + return providerOptionFunc(func(p *Provider) { p.jsonMode = m }) +} + +// WithImageInput declares whether this endpoint accepts image parts on user +// messages. Same reasoning as WithJSONMode: New reports OpenAI's own +// support, NewCompat claims none until the caller declares it. +// +// It sets Capabilities().ImageInput only. Media parts are rendered on the +// wire either way, so this changes what the provider promises, not what it +// sends — a caller that skips the capability check still reaches the +// endpoint's own rejection. +func WithImageInput(ok bool) Option { + return providerOptionFunc(func(p *Provider) { p.imageInput = ok }) +} + +// structuredOutputFor translates the provider-neutral structured-output +// request on ctx into this endpoint's JSON mode. It returns the +// response_format to send plus the messages to send with it, since +// JSONModeObject carries the schema as an extra message. Both are returned +// unchanged when no structured output was requested. +func (p *Provider) structuredOutputFor( + ctx context.Context, + msgs []openai.ChatCompletionMessage, +) (*openai.ChatCompletionResponseFormat, []openai.ChatCompletionMessage, error) { + so := agent.StructuredOutputFromContext(ctx) + if so == nil { + return nil, msgs, nil + } + switch p.jsonMode { + case JSONModeSchema: + name := so.Name + if name == "" { + name = "response" + } + return &openai.ChatCompletionResponseFormat{ + Type: openai.ChatCompletionResponseFormatTypeJSONSchema, + JSONSchema: &openai.ChatCompletionResponseFormatJSONSchema{ + Name: name, + Description: so.Description, + Schema: jsonSchemaMarshaler(so.Schema), + Strict: so.Strict, + }, + }, msgs, nil + case JSONModeObject: + instruction, err := jsonObjectInstruction(so) + if err != nil { + return nil, nil, err + } + // Appended last rather than merged into an existing system message: + // the schema is an instruction, and the final message is the one + // models follow most reliably. + return &openai.ChatCompletionResponseFormat{ + Type: openai.ChatCompletionResponseFormatTypeJSONObject, + }, append(msgs, openai.ChatCompletionMessage{ + Role: openai.ChatMessageRoleSystem, + Content: instruction, + }), nil + default: + return nil, nil, fmt.Errorf( + "openai: structured output requested but this endpoint declares %s: pass WithJSONMode(JSONModeSchema) or WithJSONMode(JSONModeObject) to the constructor", + p.jsonMode) + } +} + +// jsonObjectInstruction renders the schema as a prompt fragment, which is +// where it has to live under JSONModeObject. +// +// The literal word JSON is load-bearing, not decoration: endpoints +// implementing this format commonly reject a request whose messages never +// mention it, to stop callers from switching on JSON mode and then asking +// for prose. +func jsonObjectInstruction(so *agent.StructuredOutput) (string, error) { + schema, err := so.MarshalSchema() + if err != nil { + return "", fmt.Errorf("openai: encoding structured-output schema: %w", err) + } + var b strings.Builder + b.Grow(len(schema) + 256) + b.WriteString("Reply with a single JSON object and nothing else: no prose, no code fence.\n") + b.WriteString("It must validate against this JSON Schema:\n") + b.Write(schema) + if so.Description != "" { + b.WriteString("\nThe object represents: ") + b.WriteString(so.Description) + } + if so.Strict { + b.WriteString("\nDo not emit any property the schema does not declare.") + } + return b.String(), nil +} diff --git a/pkg/llm/openai/json_mode_test.go b/pkg/llm/openai/json_mode_test.go new file mode 100644 index 0000000..20412cc --- /dev/null +++ b/pkg/llm/openai/json_mode_test.go @@ -0,0 +1,183 @@ +package openai + +import ( + "context" + "io" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/hung12ct/gopheragent/pkg/agent" + "github.com/hung12ct/gopheragent/pkg/history" +) + +func personSchema() agent.StructuredOutput { + return agent.StructuredOutput{ + Name: "person", + Description: "a human", + Schema: map[string]any{ + "type": "object", + "properties": map[string]any{ + "name": map[string]any{"type": "string"}, + }, + "required": []string{"name"}, + }, + Strict: true, + } +} + +// Endpoints that publish only the older json_object format 400 on +// json_schema, so the schema has to move from response_format into the +// prompt while response_format drops to the shape they accept. +func TestJSONModeObject_SendsJSONObjectAndPromptsSchema(t *testing.T) { + req := captureOpenAIRequest(t, func(p *Provider) { + p.jsonMode = JSONModeObject + ctx := agent.WithStructuredOutput(context.Background(), personSchema()) + ch := make(chan agent.StreamEvent, 4) + go func() { + for range ch { + } + }() + _, _ = p.GenerateStream(ctx, []history.Message{{Role: "user", Content: "hi"}}, nil, ch) + close(ch) + }) + + rf, ok := req["response_format"].(map[string]any) + if !ok { + t.Fatalf("response_format missing or wrong shape: %v", req["response_format"]) + } + if rf["type"] != "json_object" { + t.Fatalf("type: want json_object, got %v", rf["type"]) + } + if _, present := rf["json_schema"]; present { + t.Fatalf("json_schema must not be sent in object mode: %v", rf) + } + + msgs, ok := req["messages"].([]any) + if !ok || len(msgs) != 2 { + t.Fatalf("want 2 messages (user + appended schema), got %v", req["messages"]) + } + last, _ := msgs[len(msgs)-1].(map[string]any) + if last["role"] != "system" { + t.Fatalf("appended message role = %v, want system", last["role"]) + } + content, _ := last["content"].(string) + if !strings.Contains(content, `"name"`) || !strings.Contains(content, "object") { + t.Fatalf("appended message must carry the schema, got %q", content) + } + // Endpoints in this mode commonly reject a request whose messages never + // say JSON, so the rendered instruction has to contain the literal word. + if !strings.Contains(content, "JSON") { + t.Fatalf("appended message must contain the word JSON, got %q", content) + } + if !strings.Contains(content, "a human") { + t.Fatalf("appended message should carry the schema description, got %q", content) + } +} + +// The prompt-injected schema is a per-request concern; leaking it into the +// conversation the caller passed would repeat it on every later turn. +func TestJSONModeObject_DoesNotMutateCallerMessages(t *testing.T) { + memory := []history.Message{{Role: "user", Content: "hi"}} + captureOpenAIRequest(t, func(p *Provider) { + p.jsonMode = JSONModeObject + ctx := agent.WithStructuredOutput(context.Background(), personSchema()) + ch := make(chan agent.StreamEvent, 4) + go func() { + for range ch { + } + }() + _, _ = p.GenerateStream(ctx, memory, nil, ch) + close(ch) + }) + if len(memory) != 1 { + t.Fatalf("caller history mutated: len = %d, want 1", len(memory)) + } +} + +// Object mode still must not send response_format on an ordinary call. +func TestJSONModeObject_NoStructuredOutputOmitsResponseFormat(t *testing.T) { + req := captureOpenAIRequest(t, func(p *Provider) { + p.jsonMode = JSONModeObject + ch := make(chan agent.StreamEvent, 4) + go func() { + for range ch { + } + }() + _, _ = p.GenerateStream(context.Background(), []history.Message{{Role: "user", Content: "hi"}}, nil, ch) + close(ch) + }) + if _, ok := req["response_format"]; ok { + t.Fatalf("response_format must be omitted without a schema, got %v", req["response_format"]) + } +} + +// Failing loudly beats downgrading to prose the caller will try to +// unmarshal, and beats a 400 whose message names neither the cause nor the +// fix. The request must not reach the endpoint at all. +func TestJSONModeNone_FailsStructuredOutputBeforeSending(t *testing.T) { + var reached bool + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + reached = true + w.Header().Set("Content-Type", "text/event-stream") + _, _ = io.WriteString(w, "data: [DONE]\n\n") + })) + t.Cleanup(srv.Close) + + p, err := NewCompat("test-key", "test/model", srv.URL+"/v1") + if err != nil { + t.Fatalf("NewCompat: %v", err) + } + ctx := agent.WithStructuredOutput(context.Background(), personSchema()) + ch := make(chan agent.StreamEvent, 4) + go func() { + for range ch { + } + }() + _, err = p.GenerateStream(ctx, []history.Message{{Role: "user", Content: "hi"}}, nil, ch) + close(ch) + + if reached { + t.Fatal("request must not be sent when no JSON mode is declared") + } + if err == nil { + t.Fatal("want an error when structured output is requested with JSONModeNone") + } + if !strings.Contains(err.Error(), "WithJSONMode") { + t.Fatalf("error must name the fix, got %v", err) + } +} + +// Without a schema on the context, a JSONModeNone provider is an ordinary +// chat provider and must keep working. +func TestJSONModeNone_AllowsPlainChat(t *testing.T) { + req := captureOpenAIRequest(t, func(p *Provider) { + p.jsonMode = JSONModeNone + ch := make(chan agent.StreamEvent, 4) + go func() { + for range ch { + } + }() + if _, err := p.GenerateStream(context.Background(), []history.Message{{Role: "user", Content: "hi"}}, nil, ch); err != nil { + t.Errorf("plain chat must not fail under JSONModeNone: %v", err) + } + close(ch) + }) + if _, ok := req["response_format"]; ok { + t.Fatalf("response_format must be omitted, got %v", req["response_format"]) + } +} + +func TestJSONModeString(t *testing.T) { + for mode, want := range map[JSONMode]string{ + JSONModeSchema: "JSONModeSchema", + JSONModeObject: "JSONModeObject", + JSONModeNone: "JSONModeNone", + JSONMode(9): "JSONMode(9)", + } { + if got := mode.String(); got != want { + t.Fatalf("String() = %q, want %q", got, want) + } + } +} diff --git a/pkg/llm/openai/openai.go b/pkg/llm/openai/openai.go index d512b02..1108b5c 100644 --- a/pkg/llm/openai/openai.go +++ b/pkg/llm/openai/openai.go @@ -37,6 +37,8 @@ type Provider struct { temperature *float64 topP *float64 seed *int64 + jsonMode JSONMode + imageInput bool cfg clientConfig } @@ -63,10 +65,19 @@ func WithSeed(n int64) Option { return providerOptionFunc(func(p *Provider) { p.seed = &n }) } -// Capabilities reports features implemented by this adapter. Compatible -// endpoints still need model-specific discovery when their catalog varies. -func (*Provider) Capabilities() agent.LLMCapabilities { - return agent.LLMCapabilities{ImageInput: true, StructuredOutput: true} +// Capabilities reports what this provider's endpoint accepts, as configured. +// +// New reports api.openai.com's features. NewCompat claims nothing until the +// caller declares its endpoint with WithJSONMode / WithImageInput: a +// compatible endpoint implements an unknown subset, and the whole value of +// this signal is that a consumer can reject an unsuitable provider at +// construction — a blanket claim on behalf of every gateway in the ecosystem +// would hand that consumer a confident wrong answer instead. +func (p *Provider) Capabilities() agent.LLMCapabilities { + return agent.LLMCapabilities{ + ImageInput: p.imageInput, + StructuredOutput: p.jsonMode != JSONModeNone, + } } // resolveAPIKey falls back to OPENAI_API_KEY when apiKey is empty. @@ -90,7 +101,9 @@ func New(apiKey string, model string, opts ...Option) (*Provider, error) { if model == "" { model = openai.GPT4o } - p := &Provider{model: model} + // jsonMode's zero value is JSONModeSchema, which is what api.openai.com + // implements; NewCompat prepends overrides for both defaults. + p := &Provider{model: model, imageInput: true} for _, opt := range opts { opt.applyProvider(p) } @@ -187,11 +200,17 @@ func (p *Provider) GenerateStream(ctx context.Context, memory []history.Message, } } + respFormat, reqMessages, err := p.structuredOutputFor(ctx, reqMessages) + if err != nil { + return agent.LLMResult{}, err + } + req := openai.ChatCompletionRequest{ - Model: p.model, - Messages: reqMessages, - Tools: openaiTools, - Stream: true, + Model: p.model, + Messages: reqMessages, + Tools: openaiTools, + ResponseFormat: respFormat, + Stream: true, StreamOptions: &openai.StreamOptions{ IncludeUsage: true, }, @@ -200,22 +219,6 @@ func (p *Provider) GenerateStream(ctx context.Context, memory []history.Message, if effort := reasoningEffortFor(p.model, agent.ThinkingBudgetFromContext(ctx)); effort != "" { req.ReasoningEffort = effort } - if so := agent.StructuredOutputFromContext(ctx); so != nil { - name := so.Name - if name == "" { - name = "response" - } - req.ResponseFormat = &openai.ChatCompletionResponseFormat{ - Type: openai.ChatCompletionResponseFormatTypeJSONSchema, - JSONSchema: &openai.ChatCompletionResponseFormatJSONSchema{ - Name: name, - Description: so.Description, - Schema: jsonSchemaMarshaler(so.Schema), - Strict: so.Strict, - }, - } - } - stream, err := p.client.CreateChatCompletionStream(ctx, req) if err != nil { return agent.LLMResult{}, fmt.Errorf("openai streaming error: %w", classifyErr(err)) diff --git a/pkg/llm/openai/openai_compat.go b/pkg/llm/openai/openai_compat.go index 31fa25d..71c3863 100644 --- a/pkg/llm/openai/openai_compat.go +++ b/pkg/llm/openai/openai_compat.go @@ -17,12 +17,30 @@ import ( // point NewEmbedder, NewVisionAnalyzer, and NewSummaryProvider at the same // endpoint with WithBaseURL, or they will call api.openai.com. // +// Two features default off here for the same reason the model is required: +// a compatible endpoint implements an unknown subset of the API, so the +// adapter declares nothing it cannot know. +// +// - JSON mode is JSONModeNone. A structured-output call fails until +// WithJSONMode says what the endpoint implements. Endpoints publishing +// OpenAI's json_schema take JSONModeSchema; those publishing only the +// older json_object take JSONModeObject. +// - Image input is unclaimed until WithImageInput(true). +// +// Both feed Capabilities(), so a consumer that needs schema enforcement or +// vision can reject the provider at construction instead of at the first +// request. +// // Examples: // -// Gemini: NewCompat("GEMINI_API_KEY", "gemini-2.0-flash", "https://generativelanguage.googleapis.com/v1beta/openai") -// OpenRouter: NewCompat("OPENROUTER_API_KEY", "openai/gpt-4o", "https://openrouter.ai/api/v1") -// Ollama: NewCompat("ollama", "llama3", "http://localhost:11434/v1") -// Groq: NewCompat("GROQ_KEY", "llama-3.3-70b-versatile", "https://api.groq.com/openai/v1") +// Gemini: NewCompat("GEMINI_API_KEY", "gemini-2.0-flash", "https://generativelanguage.googleapis.com/v1beta/openai", +// WithJSONMode(JSONModeSchema), WithImageInput(true)) +// OpenRouter: NewCompat("OPENROUTER_API_KEY", "openai/gpt-4o", "https://openrouter.ai/api/v1", +// WithJSONMode(JSONModeSchema), WithImageInput(true)) +// DeepSeek: NewCompat("DEEPSEEK_API_KEY", "deepseek-v4-flash", "https://api.deepseek.com/v1", +// WithJSONMode(JSONModeObject)) +// Ollama: NewCompat("ollama", "llama3", "http://localhost:11434/v1", WithJSONMode(JSONModeObject)) +// Groq: NewCompat("GROQ_KEY", "llama-3.3-70b-versatile", "https://api.groq.com/openai/v1", WithJSONMode(JSONModeObject)) func NewCompat(apiKey string, model string, baseURL string, opts ...Option) (*Provider, error) { if resolveAPIKey(apiKey) == "" { return nil, errors.New("openai: NewCompat: API key is not set") @@ -35,5 +53,12 @@ func NewCompat(apiKey string, model string, baseURL string, opts ...Option) (*Pr if _, err := validateBaseURL(baseURL); err != nil { return nil, fmt.Errorf("openai: NewCompat: %w", err) } - return New(apiKey, model, append([]Option{WithBaseURL(baseURL)}, opts...)...) + // Prepended, not appended: these downgrade New's OpenAI-shaped defaults + // to "unknown endpoint", and a caller's own opts must still win. + defaults := []Option{ + WithBaseURL(baseURL), + WithJSONMode(JSONModeNone), + WithImageInput(false), + } + return New(apiKey, model, append(defaults, opts...)...) } diff --git a/pkg/llm/openai/openai_compat_test.go b/pkg/llm/openai/openai_compat_test.go index c6850bb..8956c1a 100644 --- a/pkg/llm/openai/openai_compat_test.go +++ b/pkg/llm/openai/openai_compat_test.go @@ -46,8 +46,57 @@ func TestNewCompatSendsConfiguredHeaders(t *testing.T) { } func TestOpenAIReportsMultimodalStructuredTransport(t *testing.T) { - caps := (&Provider{}).Capabilities() + p, err := New("test-key", "gpt-4o") + if err != nil { + t.Fatalf("New: %v", err) + } + caps := p.Capabilities() if !caps.ImageInput || !caps.StructuredOutput { t.Fatalf("Capabilities = %+v, want image input and structured output", caps) } } + +// A compatible endpoint implements an unknown subset, so NewCompat must not +// claim OpenAI's features on its behalf: a consumer that checks capabilities +// to reject an unsuitable provider gets a wrong answer at construction and +// discovers the truth as a 400 mid-run. +func TestNewCompatClaimsNothingUntilDeclared(t *testing.T) { + p, err := NewCompat("test-key", "test/model", "https://api.example.com/v1") + if err != nil { + t.Fatalf("NewCompat: %v", err) + } + if caps := p.Capabilities(); caps.ImageInput || caps.StructuredOutput { + t.Fatalf("Capabilities = %+v, want no claims by default", caps) + } +} + +func TestNewCompatCapabilitiesFollowDeclaration(t *testing.T) { + for _, tc := range []struct { + name string + opts []Option + want bool + }{ + {"schema", []Option{WithJSONMode(JSONModeSchema)}, true}, + {"object", []Option{WithJSONMode(JSONModeObject)}, true}, + {"none", []Option{WithJSONMode(JSONModeNone)}, false}, + } { + t.Run(tc.name, func(t *testing.T) { + p, err := NewCompat("test-key", "test/model", "https://api.example.com/v1", tc.opts...) + if err != nil { + t.Fatalf("NewCompat: %v", err) + } + if got := p.Capabilities().StructuredOutput; got != tc.want { + t.Fatalf("StructuredOutput = %v, want %v", got, tc.want) + } + }) + } + + // Caller options must beat the defaults NewCompat prepends. + p, err := NewCompat("test-key", "test/model", "https://api.example.com/v1", WithImageInput(true)) + if err != nil { + t.Fatalf("NewCompat: %v", err) + } + if !p.Capabilities().ImageInput { + t.Fatal("WithImageInput(true) must override the NewCompat default") + } +}