Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
fix(provider): opencode-go context limits and 400 error surfacing (#5…
  • Loading branch information
JerryLiu369 committed Oct 2, 2026
commit 5290f5edf86254684f5077f1a92f3568e0fa4fe8
32 changes: 28 additions & 4 deletions packages/core/src/plugin/models-dev.ts
Original file line number Diff line number Diff line change
Expand Up @@ -73,6 +73,26 @@ function modeName(model: ModelsDev.Model, mode: string) {
return `${model.name} ${mode.charAt(0).toUpperCase()}${mode.slice(1)}`
}

// The opencode-go gateway rejects prompts around ~147-148k input tokens while
// the upstream catalog advertises up to 1M context. Without a local clamp,
// automatic compaction never fires and the session bricks on repeated HTTP
// 400s. Keep the effective limits at the real gateway constraint so both
// proactive compaction and overflow recovery engage in time.
const OPENCODE_GO_CONTEXT_LIMIT = 148_000
const OPENCODE_GO_INPUT_LIMIT = 128_000

function clampOpencodeGoLimit(providerID: ProviderV2.ID, draft: ModelV2Info) {
if (providerID !== "opencode-go") return
if (draft.limit.context <= OPENCODE_GO_CONTEXT_LIMIT && draft.limit.input === undefined) return
if (draft.limit.context > OPENCODE_GO_CONTEXT_LIMIT) draft.limit.context = OPENCODE_GO_CONTEXT_LIMIT
const input = draft.limit.input
if (input === undefined) {
if (draft.limit.context >= OPENCODE_GO_INPUT_LIMIT) draft.limit.input = OPENCODE_GO_INPUT_LIMIT
return
}
if (input > OPENCODE_GO_INPUT_LIMIT) draft.limit.input = Math.min(OPENCODE_GO_INPUT_LIMIT, draft.limit.context)
}

function applyModel(
draft: ModelV2Info,
model: ModelsDev.Model,
Expand Down Expand Up @@ -161,15 +181,19 @@ export const ModelsDevPlugin = define({

for (const model of Object.values(item.models)) {
const baseCost = cost(model.cost)
catalog.model.update(providerID, model.id, (draft) => applyModel(draft, model, { cost: baseCost }))
catalog.model.update(providerID, model.id, (draft) => {
applyModel(draft, model, { cost: baseCost })
clampOpencodeGoLimit(providerID, draft)
})
for (const [mode, options] of Object.entries(model.experimental?.modes ?? {})) {
catalog.model.update(providerID, `${model.id}-${mode}`, (draft) =>
catalog.model.update(providerID, `${model.id}-${mode}`, (draft) => {
applyModel(draft, model, {
name: modeName(model, mode),
cost: mergeCost(baseCost, options.cost),
request: options.provider,
}),
)
})
clampOpencodeGoLimit(providerID, draft)
})
}
}
}
Expand Down
6 changes: 5 additions & 1 deletion packages/core/src/session/runner/model.ts
Original file line number Diff line number Diff line change
Expand Up @@ -92,12 +92,16 @@ const withDefaults = (model: ModelV2.Info, route: AnyRoute) => {
const httpBody = Object.hasOwn(body, "apiKey")
? Object.fromEntries(Object.entries(body).filter(([key]) => key !== "apiKey"))
: body
// Defense in depth for the opencode-go gateway (~148k input) when the
// catalog still advertises 1M context (custom configs, stale cache).
const context =
model.providerID === "opencode-go" ? Math.min(model.limit.context, 148_000) : model.limit.context
return route.with({
provider: model.providerID,
endpoint: model.api.url === undefined ? undefined : { baseURL: model.api.url },
headers: model.request.headers,
http: { body: httpBody },
limits: { context: model.limit.context, output: model.limit.output },
limits: { context, output: model.limit.output },
})
}

Expand Down
83 changes: 83 additions & 0 deletions packages/core/test/plugin/models-dev.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -124,6 +124,89 @@ describe("ModelsDevPlugin", () => {
}),
)

it.effect("clamps opencode-go limits to gateway constraints", () =>
Effect.gen(function* () {
const integrations = yield* Integration.Service
const catalog = yield* Catalog.Service
const models = ModelsDev.Service.of({
get: () =>
Effect.succeed({
"opencode-go": {
id: "opencode-go",
name: "OpenCode Go",
env: [],
npm: "@ai-sdk/openai-compatible",
api: "https://opencode.ai/zen/go/v1",
models: {
"qwen3.8-flash": {
id: "qwen3.8-flash",
name: "Qwen 3.8 Flash",
family: "qwen",
release_date: "2026-01-01",
attachment: false,
reasoning: false,
temperature: true,
tool_call: true,
limit: { context: 1_000_000, output: 65_536 },
},
"small-model": {
id: "small-model",
name: "Small",
family: "small",
release_date: "2026-01-01",
attachment: false,
reasoning: false,
temperature: true,
tool_call: true,
limit: { context: 100_000, output: 8_192 },
},
},
},
acme: {
id: "acme",
name: "Acme",
env: [],
npm: "@ai-sdk/openai-compatible",
api: "https://api.acme.test/v1",
models: {
large: {
id: "large",
name: "Large",
family: "large",
release_date: "2026-01-01",
attachment: false,
reasoning: false,
temperature: true,
tool_call: true,
limit: { context: 1_000_000, output: 8_192 },
},
},
},
} satisfies Record<string, ModelsDev.Provider>),
refresh: () => Effect.void,
})

yield* ModelsDevPlugin.effect(
host({
catalog: catalogHost(catalog),
integration: integrationHost(integrations),
}),
).pipe(Effect.provideService(ModelsDev.Service, models))

const goProvider = ProviderV2.ID.make("opencode-go")
const flashed = yield* catalog.model.get(goProvider, ModelV2.ID.make("qwen3.8-flash"))
expect(flashed?.limit.context).toBe(148_000)
expect(flashed?.limit.input).toBe(128_000)

const small = yield* catalog.model.get(goProvider, ModelV2.ID.make("small-model"))
expect(small?.limit.context).toBe(100_000)
expect(small?.limit.input).toBeUndefined()

const acmeLarge = yield* catalog.model.get(ProviderV2.ID.make("acme"), ModelV2.ID.make("large"))
expect(acmeLarge?.limit.context).toBe(1_000_000)
}),
)

it.effect("registers key methods for providers with environment variables", () =>
Effect.acquireUseRelease(
Effect.sync(() => {
Expand Down
13 changes: 13 additions & 0 deletions packages/core/test/session-runner-model.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -344,4 +344,17 @@ describe("SessionRunnerModel", () => {
expect(SessionRunnerModel.supported(model({ type: "native", settings: {} }))).toBe(false)
}),
)

it.effect("clamps opencode-go context to the gateway limit", () =>
Effect.gen(function* () {
const catalog = ModelV2.Info.make({
...model({ type: "aisdk", package: "@ai-sdk/anthropic", url: "https://opencode.ai/zen/go/v1" }),
providerID: ProviderV2.ID.make("opencode-go"),
limit: { context: 1_000_000, output: 65_536 },
})
const resolved = yield* SessionRunnerModel.fromCatalogModel(catalog)

expect(resolved.route.defaults.limits?.context).toBe(148_000)
}),
)
})
2 changes: 1 addition & 1 deletion packages/llm/src/protocols/anthropic-messages.ts
Original file line number Diff line number Diff line change
Expand Up @@ -806,7 +806,7 @@ const onError = (state: ParserState, event: AnthropicEvent): StepResult => [
[
LLMEvent.providerError({
message: providerErrorMessage(event),
classification: isContextOverflow(event.error?.message ?? "") ? "context-overflow" : undefined,
classification: isContextOverflow(providerErrorMessage(event)) ? "context-overflow" : undefined,
}),
],
]
Expand Down
18 changes: 17 additions & 1 deletion packages/llm/src/provider-error.ts
Original file line number Diff line number Diff line change
Expand Up @@ -29,13 +29,29 @@ const patterns = [
/model_context_window_exceeded/i,
/too many tokens/i,
/token limit exceeded/i,
// Gateway-style rejections (e.g. opencode-go anthropic-messages 400s) that
// mention the offending input/prompt without using the exact upstream
// phrasing above. Each pattern requires an overflow signal (exceed, too
// long/large, maximum, limit) alongside the input scope so generic 400s like
// "invalid parameter" stay non-overflow.
/prompt.*exceed/i,
/input.*exceed/i,
/context.*exceed/i,
/exceed.*context/i,
/input.*too long/i,
/input.*too large/i,
/prompt.*too (long|large)/i,
/maximum.*input.*length/i,
/token.*limit.*exceed|exceed.*token.*limit/i,
/maximum.*tokens?.*exceed|exceed.*maximum.*tokens?/i,
/exceed.*\d[\d,]*\s*tokens?/i,
]

const exclusions = [/^(throttling error|service unavailable):/i, /rate limit/i, /too many requests/i]

export const isContextOverflow = (message: string) =>
!exclusions.some((pattern) => pattern.test(message)) &&
(patterns.some((pattern) => pattern.test(message)) || /^4(00|13)\s*(status code)?\s*\(no body\)/i.test(message))
(patterns.some((pattern) => pattern.test(message)) || /4(00|13)\s*(status code)?\s*\(no body\)/i.test(message))

export const isContextOverflowFailure = (failure: unknown) =>
failure instanceof LLMError
Expand Down
11 changes: 8 additions & 3 deletions packages/llm/src/route/executor.ts
Original file line number Diff line number Diff line change
Expand Up @@ -201,9 +201,13 @@ const responseBody = (body: string | void, request: HttpClientRequest.HttpClient
return { body: redacted.slice(0, BODY_LIMIT), bodyTruncated: true }
}

const MESSAGE_BODY_LIMIT = 2_000

const providerMessage = (status: number, body: { readonly body?: string }) => {
if (body.body && body.body.length <= 500) return `Provider request failed with HTTP ${status}: ${body.body}`
return `Provider request failed with HTTP ${status}`
if (!body.body) return `Provider request failed with HTTP ${status} (no body)`
if (body.body.length <= MESSAGE_BODY_LIMIT) return `Provider request failed with HTTP ${status}: ${body.body}`
const truncated = body.body.slice(0, MESSAGE_BODY_LIMIT)
return `Provider request failed with HTTP ${status}: ${truncated}… [truncated ${body.body.length - MESSAGE_BODY_LIMIT} chars]`
}

const responseHttp = (input: {
Expand Down Expand Up @@ -259,7 +263,8 @@ const statusReason = (input: {
) {
return new InvalidRequestReason({
message: input.message,
classification: isContextOverflow(body) ? "context-overflow" : undefined,
classification:
isContextOverflow(body) || isContextOverflow(input.message) ? "context-overflow" : undefined,
http: input.http,
})
}
Expand Down
102 changes: 102 additions & 0 deletions packages/llm/test/opencode-go-gateway.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,102 @@
import { describe, expect } from "bun:test"
import { Effect, Layer, Ref } from "effect"
import { Headers, HttpClient, HttpClientRequest, HttpClientResponse } from "effect/unstable/http"
import { LLMError } from "../src"
import { RequestExecutor } from "../src/route"
import { it } from "./lib/effect"

const request = HttpClientRequest.post("https://opencode.ai/zen/go/v1/messages").pipe(
HttpClientRequest.setHeaders(Headers.fromInput({ "x-api-key": "test" })),
)

const responsesLayer = (responses: ReadonlyArray<Response>) =>
RequestExecutor.layer.pipe(
Layer.provide(
Layer.unwrap(
Effect.gen(function* () {
const cursor = yield* Ref.make(0)
return Layer.succeed(
HttpClient.HttpClient,
HttpClient.make((req) =>
Effect.gen(function* () {
const index = yield* Ref.getAndUpdate(cursor, (value) => value + 1)
return HttpClientResponse.fromWeb(req, responses[index] ?? responses[responses.length - 1])
}),
),
)
}),
),
),
)

const expectLLMError = (error: unknown) => {
expect(error).toBeInstanceOf(LLMError)
if (!(error instanceof LLMError)) throw new Error("expected LLMError")
return error
}

const httpBody = (error: LLMError) => ("http" in error.reason ? error.reason.http?.body : undefined)

describe("opencode-go gateway 400", () => {
it.effect("classifies gateway input-limit rejection as context overflow and surfaces the body", () =>
Effect.gen(function* () {
const executor = yield* RequestExecutor.Service
const error = yield* executor.execute(request).pipe(Effect.flip)

expectLLMError(error)
expect(error.reason).toMatchObject({ _tag: "InvalidRequest", classification: "context-overflow" })
expect(error.message).toContain("Input exceeds maximum input length of 148000 tokens")
expect(httpBody(error)).toContain("Input exceeds maximum input length")
}).pipe(
Effect.provide(
responsesLayer([
new Response(
JSON.stringify({
error: { type: "invalid_request_error", message: "Input exceeds maximum input length of 148000 tokens" },
}),
{ status: 400 },
),
]),
),
),
)

it.effect("surfaces long rejection bodies instead of returning a bare status message", () =>
Effect.gen(function* () {
const executor = yield* RequestExecutor.Service
const error = yield* executor.execute(request).pipe(Effect.flip)

expectLLMError(error)
expect(error.reason).toMatchObject({ _tag: "InvalidRequest", classification: "context-overflow" })
expect(error.message).toContain("HTTP 400")
expect(error.message).not.toBe("RequestExecutor.execute: Provider request failed with HTTP 400")
expect(httpBody(error)).toContain("Input exceeds 148000 tokens")
}).pipe(
Effect.provide(
responsesLayer([
new Response(
JSON.stringify({
error: {
type: "invalid_request_error",
message: `Input exceeds 148000 tokens: ${"x".repeat(800)}`,
},
}),
{ status: 400 },
),
]),
),
),
)

it.effect("surfaces non-overflow 400 bodies without misclassifying them", () =>
Effect.gen(function* () {
const executor = yield* RequestExecutor.Service
const error = yield* executor.execute(request).pipe(Effect.flip)

expectLLMError(error)
expect(error.reason).toMatchObject({ _tag: "InvalidRequest" })
expect("classification" in error.reason ? error.reason.classification : undefined).toBeUndefined()
expect(error.message).toContain("invalid parameter")
}).pipe(Effect.provide(responsesLayer([new Response("invalid parameter", { status: 400 })]))),
)
})
25 changes: 25 additions & 0 deletions packages/llm/test/provider-error.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,31 @@ describe("provider error classification", () => {
expect(messages.every(isContextOverflow)).toBe(true)
})

test("classifies opencode-go gateway limit messages as context overflow", () => {
const messages = [
"Input exceeds maximum input length of 148000 tokens",
"Prompt exceeds token limit: 150000 > 148000",
"Input exceeds 148000 tokens",
"invalid_request_error: prompt is too long: 210000 tokens",
"Provider request failed with HTTP 400: Input exceeds maximum input length of 148000 tokens",
"Provider request failed with HTTP 400 (no body)",
"400 status code (no body)",
]

expect(messages.every(isContextOverflow)).toBe(true)
})

test("does not classify generic invalid requests as context overflow", () => {
const messages = [
"invalid parameter",
"Provider request failed with HTTP 400: invalid parameter",
"request too large",
"Provider request failed with HTTP 413: request too large",
]

expect(messages.some(isContextOverflow)).toBe(false)
})

test("does not classify rate limits as context overflow", () => {
const messages = [
"Throttling error: Too many tokens, please wait before trying again.",
Expand Down
Loading
Loading