Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions .github/workflows/test.yml
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,10 @@ jobs:
uses: golangci/golangci-lint-action@v9
with:
version: v2.6
# Temporary workaround: config verification fetches the remote JSON schema and
# intermittently times out in CI before lint runs. Re-enable by 2026-04-01 once
# the schema fetch is reliable again; track in this PR/branch discussion.
verify: false
Comment thread
coderabbitai[bot] marked this conversation as resolved.

test-unit:
name: Unit Tests
Expand Down
3 changes: 1 addition & 2 deletions GETTING_STARTED.md
Original file line number Diff line number Diff line change
Expand Up @@ -613,9 +613,8 @@ console.log(embedding.data[0].embedding.slice(0, 5)); // first 5 dimensions

1. **Model routing**: The gateway automatically routes requests to the correct provider based on the model name — no configuration needed. Just use any model name from the list above.
2. **API compatibility**: The gateway exposes an OpenAI-compatible API. Existing OpenAI client libraries work unchanged for all providers.
3. **Streaming**: All providers support streaming. The gateway normalises provider-specific formats to OpenAI's SSE format.
3. **Streaming**: All providers support streaming. SSE chunks are flushed incrementally, and streaming responses terminate with `data: [DONE]`.
4. **System messages**: Anthropic's system message format is handled automatically. Gemini uses Google's OpenAI-compatible endpoint, which also handles system messages natively.
5. **Max tokens**: Anthropic requires `max_tokens` to be set. If not provided, the gateway defaults to 4096. OpenAI and Gemini treat it as optional.
6. **Responses API**: The `/v1/responses` endpoint provides a unified interface across all providers. Providers that do not natively support the Responses API convert requests internally.
7. **Embeddings**: The `/v1/embeddings` endpoint is supported by OpenAI, Gemini, Groq, xAI, and Ollama. Anthropic does not offer embeddings natively.

13 changes: 11 additions & 2 deletions internal/providers/openai/openai.go
Original file line number Diff line number Diff line change
Expand Up @@ -187,13 +187,22 @@ func (p *Provider) Responses(ctx context.Context, req *core.ResponsesRequest) (*
return &resp, nil
}

// StreamResponses returns a raw response body for streaming Responses API (caller must close)
// StreamResponses returns a normalized streaming Responses API body.
// The returned io.ReadCloser is wrapped by providers.EnsureResponsesDone, so
// callers must not assume it contains verbatim upstream bytes; the wrapper may
// synthesize a terminal `data: [DONE]` marker on completed streams. Callers
// remain responsible for closing the returned stream.
func (p *Provider) StreamResponses(ctx context.Context, req *core.ResponsesRequest) (io.ReadCloser, error) {
return p.client.DoStream(ctx, llmclient.Request{
stream, err := p.client.DoStream(ctx, llmclient.Request{
Method: http.MethodPost,
Endpoint: "/responses",
Body: req.WithStreaming(),
})
if err != nil {
return nil, err
}

return providers.EnsureResponsesDone(stream), nil
}

// Embeddings sends an embeddings request to OpenAI
Expand Down
6 changes: 2 additions & 4 deletions internal/providers/openai/openai_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -623,8 +623,6 @@ data: {"type":"response.output_text.delta","delta":"!"}

event: response.completed
data: {"type":"response.completed","response":{"id":"resp_123","object":"response","status":"completed","model":"gpt-4o"}}

data: [DONE]
`,
expectedError: false,
checkStream: func(t *testing.T, body io.ReadCloser) {
Expand Down Expand Up @@ -952,8 +950,8 @@ data: [DONE]
provider.SetBaseURL(server.URL)

req := &core.ChatRequest{
Model: "o4-mini",
Messages: []core.Message{{Role: "user", Content: "Hello"}},
Model: "o4-mini",
Messages: []core.Message{{Role: "user", Content: "Hello"}},
MaxTokens: &maxTokens,
}

Expand Down
201 changes: 201 additions & 0 deletions internal/providers/responses_done_wrapper.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,201 @@
package providers

import (
"bytes"
"io"
)

var responsesDoneMarker = []byte("data: [DONE]\n\n")

var responsesDoneLine = []byte("data: [DONE]")

var responsesDataPrefix = []byte("data: ")

var responsesCompletionPatterns = [][]byte{
[]byte(`"type":"response.completed"`),
[]byte(`"type":"response.done"`),
}

// EnsureResponsesDone normalizes Responses API streams so clients always receive
// a terminal data: [DONE] marker when the upstream stream reaches a completed
// Responses event but closes at EOF before sending the final marker.
func EnsureResponsesDone(stream io.ReadCloser) io.ReadCloser {
if stream == nil {
return nil
}

return &responsesDoneWrapper{
ReadCloser: stream,
atEventBoundary: true,
currentLineAtEventBoundary: true,
}
}

type responsesDoneWrapper struct {
io.ReadCloser
lineBuf []byte
pending []byte
sawDone bool
eventCompletedCandidate bool
completedEventReadyForDone bool
atEventBoundary bool
currentLineAtEventBoundary bool
emitted bool
}

func (w *responsesDoneWrapper) Read(p []byte) (int, error) {
if len(w.pending) > 0 {
n := copy(p, w.pending)
w.pending = w.pending[n:]
if len(w.pending) == 0 {
w.emitted = true
}
return n, nil
}

if w.emitted {
return 0, io.EOF
}

n, err := w.ReadCloser.Read(p)
if n > 0 {
w.trackStream(p[:n])
}

if err == io.EOF {
if w.sawDone {
if n > 0 {
return n, nil
}
return 0, io.EOF
}

missingSuffix := w.synthesizeDoneSuffix()
if len(missingSuffix) == 0 {
if n > 0 {
return n, nil
}
return 0, io.EOF
}

if n > 0 {
w.pending = append(w.pending[:0], missingSuffix...)
return n, nil
}

n = copy(p, missingSuffix)
if n < len(missingSuffix) {
w.pending = append(w.pending[:0], missingSuffix[n:]...)
return n, nil
}

w.emitted = true
return n, nil
Comment thread
coderabbitai[bot] marked this conversation as resolved.
}

return n, err
}

func (w *responsesDoneWrapper) trackStream(data []byte) {
start := 0
for i, b := range data {
if b != '\n' {
continue
}

w.lineBuf = append(w.lineBuf, data[start:i]...)
w.processLine(w.lineBuf)
w.lineBuf = w.lineBuf[:0]
start = i + 1
w.currentLineAtEventBoundary = w.atEventBoundary
}

if start < len(data) {
w.lineBuf = append(w.lineBuf, data[start:]...)
}
}

func (w *responsesDoneWrapper) processLine(line []byte) {
line = bytes.TrimSuffix(line, []byte("\r"))
if len(line) == 0 {
if w.eventCompletedCandidate {
w.completedEventReadyForDone = true
}
w.eventCompletedCandidate = false
w.atEventBoundary = true
return
}

if w.completedEventReadyForDone && (!w.currentLineAtEventBoundary || !bytes.Equal(line, responsesDoneLine)) {
w.completedEventReadyForDone = false
}

if w.currentLineAtEventBoundary && bytes.Equal(line, responsesDoneLine) {
w.sawDone = true
}

if bytes.HasPrefix(line, responsesDataPrefix) {
if isCompletedDataLine(line) {
w.eventCompletedCandidate = true
}
}

w.atEventBoundary = false
}

func (w *responsesDoneWrapper) synthesizeDoneSuffix() []byte {
if w.sawDone {
return nil
}

if w.eventCompletedCandidate && len(w.lineBuf) == 0 {
return append([]byte{'\n'}, responsesDoneMarker...)
}

if isCompletedDataLine(w.lineBuf) {
return append([]byte("\n\n"), responsesDoneMarker...)
}

if !w.completedEventReadyForDone {
return nil
}

if len(w.lineBuf) == 0 {
if w.atEventBoundary {
return append([]byte(nil), responsesDoneMarker...)
}
return nil
}

if !w.currentLineAtEventBoundary || !isDoneLinePrefix(w.lineBuf) {
return nil
}

suffix := append([]byte(nil), responsesDoneLine[len(w.lineBuf):]...)
suffix = append(suffix, '\n', '\n')
return suffix
}

func isDoneLinePrefix(line []byte) bool {
if len(line) > len(responsesDoneLine) {
return false
}

return bytes.Equal(line, responsesDoneLine[:len(line)])
}

func isCompletedDataLine(line []byte) bool {
line = bytes.TrimSuffix(line, []byte("\r"))
if !bytes.HasPrefix(line, responsesDataPrefix) {
return false
}

payload := line[len(responsesDataPrefix):]
for _, pattern := range responsesCompletionPatterns {
if bytes.Contains(payload, pattern) {
return true
}
}

return false
}
Loading