From f9fb9bb9df07282b86a3a426d8f6b6b0a9657de2 Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:46:09 +0000 Subject: [PATCH 1/7] feat(api): the /v1/chat/completions contract and its wire conversion MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit First half of the route from docs/LOCAL-ENGINE-API.md: the declared contract and the translation layer. No handler yet, so nothing is served — the endpoint is not reachable until the following commit wires it. The group goes on the v2 protocol surface rather than the instance httpapi: Authorization is declared once at that HttpApi level so a new group gets it instead of having to remember .middleware(Authorization), it has the typed SSE helper (HttpApiSchema.StreamSse), and it is the surface with a contract changelog. Paths there are literal, so /v1/... sits alongside /api/... without a prefix mechanism. Tool ownership is decided by the request and nothing else: `tools` present means the caller executes them and the turn ends with finish_reason tool_calls; `tools` absent means a plain completion. One rule rather than three configurable ones, and it is what Office and browser already assume — their tools mutate an open document and a live tab, which the engine cannot do on their behalf. Errors are one class per status sharing the OpenAI envelope. Collapsing them would report a rejected key as a bad request, and the status is what a client's transport sees first. 401 carries code `engine_not_authenticated`, which is the only condition a client may turn into a sign-in prompt. The conversion layer is pure — no Effect, no services — because it is where the silent failures live: a mistranslation does not fail the request, it just gives the model something other than what the caller wrote. What it refuses is the interesting part. Malformed tool_call arguments and an unsupported content part are rejected rather than forwarded as {} or dropped, since an image quietly discarded looks to a user like the model ignored it. A tool message with no tool_call_id is refused because the provider pairs results to calls by that id and there is nothing sensible to guess. Two asymmetries the tests pin down: the wire carries tool arguments as a JSON string while the engine wants them parsed, so both directions convert; and "error"/"unknown" finish reasons have no OpenAI equivalent and must not be reported as "stop", which would tell the caller a failed turn ended cleanly. 22 tests pass. protocol and server typecheck clean. --- .../protocol/src/groups/chat-completion.ts | 203 ++++++++++++++++ .../test/server/chat-completion-wire.test.ts | 228 ++++++++++++++++++ .../src/handlers/chat-completion-wire.ts | 207 ++++++++++++++++ 3 files changed, 638 insertions(+) create mode 100644 packages/protocol/src/groups/chat-completion.ts create mode 100644 packages/redrob/test/server/chat-completion-wire.test.ts create mode 100644 packages/server/src/handlers/chat-completion-wire.ts diff --git a/packages/protocol/src/groups/chat-completion.ts b/packages/protocol/src/groups/chat-completion.ts new file mode 100644 index 0000000000..7a4a45a83a --- /dev/null +++ b/packages/protocol/src/groups/chat-completion.ts @@ -0,0 +1,203 @@ +/** + * `POST /v1/chat/completions` — the OpenAI-compatible inference route. + * + * This is the one surface every Redrob product is meant to reach the engine + * through: it carries the engine's own Console credential, so a product never + * holds a key of its own, and it resolves the same provider registry the agent + * does, so a locally-served model is reachable through the same call. + * + * It is NOT the agent. A chat completion is stateless and a session is not, so + * this route makes one model call and returns; mapping it onto the session API + * is what produced the broken half-implementations this replaces (see + * docs/LOCAL-ENGINE-API.md). + * + * WHO EXECUTES TOOLS IS DECIDED BY THE REQUEST, and that is the whole contract: + * + * - `tools` present -> the CALLER owns them. Nothing is executed here; the + * turn ends with `finish_reason: "tool_calls"` and the caller sends the + * results back as `role: "tool"` messages. Standard OpenAI semantics. + * - `tools` absent -> a plain completion. + * + * Hosts whose tools mutate something they hold — an open document, a live + * browser tab — cannot hand execution to the engine, so caller ownership is a + * requirement for them rather than a preference. + * + * The path is versioned, not the engine: fields may be ADDED here, but removing + * one or changing its meaning requires /v2. Products negotiate with + * /api/capabilities instead of pinning an engine version. + */ +import { Schema } from "effect" +import { HttpApiEndpoint, HttpApiGroup, OpenApi } from "effect/unstable/httpapi" + +/** OpenAI puts the function under a `function` key and tags the wrapper. */ +const FunctionTool = Schema.Struct({ + type: Schema.Literal("function"), + function: Schema.Struct({ + name: Schema.String, + description: Schema.optional(Schema.String), + /** Raw JSON Schema, passed to the provider untouched. */ + parameters: Schema.optional(Schema.Record(Schema.String, Schema.Unknown)), + }), +}) + +const ToolCall = Schema.Struct({ + id: Schema.String, + type: Schema.Literal("function"), + function: Schema.Struct({ + name: Schema.String, + /** A JSON *string*, which is what OpenAI clients parse. */ + arguments: Schema.String, + }), +}) + +/** + * Message content is a string in the common case and a part array when a client + * sends images. `Schema.Unknown` for the array case keeps this route from + * becoming a second, divergent definition of multimodal content — the parts are + * handed to the provider layer, which owns that shape. + */ +const MessageContent = Schema.Union([ + Schema.String, + Schema.Null, + Schema.Array(Schema.Record(Schema.String, Schema.Unknown)), +]) + +const Message = Schema.Struct({ + role: Schema.Literals(["system", "developer", "user", "assistant", "tool"]), + content: Schema.optional(MessageContent), + /** Present on an assistant message that is replaying a previous tool turn. */ + tool_calls: Schema.optional(Schema.Array(ToolCall)), + /** Required on a `tool` message: which call this is the result of. */ + tool_call_id: Schema.optional(Schema.String), + name: Schema.optional(Schema.String), +}) + +const ToolChoice = Schema.Union([ + Schema.Literals(["auto", "none", "required"]), + Schema.Struct({ + type: Schema.Literal("function"), + function: Schema.Struct({ name: Schema.String }), + }), +]) + +export const ChatCompletionRequest = Schema.Struct({ + /** `redrob/auto` routes through the Console; a local provider id stays local. */ + model: Schema.String, + messages: Schema.Array(Message), + tools: Schema.optional(Schema.Array(FunctionTool)), + tool_choice: Schema.optional(ToolChoice), + /** Default false. True returns text/event-stream. */ + stream: Schema.optional(Schema.Boolean), + max_tokens: Schema.optional(Schema.Int), + /** OpenAI's newer name for the same cap; `max_tokens` wins if both appear. */ + max_completion_tokens: Schema.optional(Schema.Int), + temperature: Schema.optional(Schema.Number), + top_p: Schema.optional(Schema.Number), + stop: Schema.optional(Schema.Union([Schema.String, Schema.Array(Schema.String)])), + seed: Schema.optional(Schema.Int), + frequency_penalty: Schema.optional(Schema.Number), + presence_penalty: Schema.optional(Schema.Number), + /** Passed through where the provider accepts it. */ + reasoning_effort: Schema.optional(Schema.String), + /** Accepted and ignored: a caller may send it, and refusing would be rude. */ + user: Schema.optional(Schema.String), +}).annotate({ identifier: "ChatCompletionRequest" }) +export type ChatCompletionRequest = typeof ChatCompletionRequest.Type + +const Usage = Schema.Struct({ + prompt_tokens: Schema.Int, + completion_tokens: Schema.Int, + total_tokens: Schema.Int, +}) + +const Choice = Schema.Struct({ + index: Schema.Int, + message: Schema.Struct({ + role: Schema.Literal("assistant"), + content: Schema.NullOr(Schema.String), + tool_calls: Schema.optional(Schema.Array(ToolCall)), + /** Present when the model emitted reasoning; not an OpenAI field. */ + reasoning_content: Schema.optional(Schema.String), + }), + finish_reason: Schema.NullOr(Schema.String), +}) + +export const ChatCompletionResponse = Schema.Struct({ + id: Schema.String, + object: Schema.Literal("chat.completion"), + created: Schema.Int, + model: Schema.String, + choices: Schema.Array(Choice), + usage: Schema.optional(Usage), +}).annotate({ identifier: "ChatCompletionResponse" }) +export type ChatCompletionResponse = typeof ChatCompletionResponse.Type + +/** + * The OpenAI error envelope, so a client's existing error handling works + * unchanged. + * + * `code` is the part clients must branch on. `engine_not_authenticated` means + * the ENGINE has no Console credential, and it is the only condition under which + * a client should offer a sign-in action — the message text is localized + * downstream and is not a contract. Branching on text instead is how a network + * timeout reached a user as "please log in". + * + * One class per status, because the status is what a client's transport sees + * first and collapsing them would report a rejected key as a bad request. The + * body shape is identical across all of them. + */ +const errorFields = { + error: Schema.Struct({ + message: Schema.String, + type: Schema.String, + code: Schema.NullOr(Schema.String), + param: Schema.optional(Schema.NullOr(Schema.String)), + }), +} + +export class ChatBadRequestError extends Schema.ErrorClass("ChatBadRequestError")( + errorFields, + { httpApiStatus: 400 }, +) {} + +/** Rejected or absent credentials. Carries code `engine_not_authenticated`. */ +export class ChatUnauthorizedError extends Schema.ErrorClass("ChatUnauthorizedError")( + errorFields, + { httpApiStatus: 401 }, +) {} + +export class ChatRateLimitError extends Schema.ErrorClass("ChatRateLimitError")( + errorFields, + { httpApiStatus: 429 }, +) {} + +/** Upstream failed, or no route could be resolved for the requested model. */ +export class ChatUpstreamError extends Schema.ErrorClass("ChatUpstreamError")( + errorFields, + { httpApiStatus: 502 }, +) {} + +export const ChatCompletionGroup = HttpApiGroup.make("server.chat") + .add( + HttpApiEndpoint.post("chat.completions", "/v1/chat/completions", { + payload: ChatCompletionRequest, + success: ChatCompletionResponse, + error: [ChatBadRequestError, ChatUnauthorizedError, ChatRateLimitError, ChatUpstreamError], + }).annotateMerge( + OpenApi.annotations({ + identifier: "v1.chat.completions", + summary: "Create a chat completion", + description: + "OpenAI-compatible inference against the engine's own credential and provider registry. " + + "Declaring `tools` makes the caller responsible for executing them: the turn ends with " + + "finish_reason tool_calls and the results come back as role:tool messages. " + + "`stream: true` returns text/event-stream instead of this JSON body.", + }), + ), + ) + .annotateMerge( + OpenApi.annotations({ + title: "chat completions", + description: "OpenAI-compatible inference route. The shared engine surface for Redrob products.", + }), + ) diff --git a/packages/redrob/test/server/chat-completion-wire.test.ts b/packages/redrob/test/server/chat-completion-wire.test.ts new file mode 100644 index 0000000000..d3bf755954 --- /dev/null +++ b/packages/redrob/test/server/chat-completion-wire.test.ts @@ -0,0 +1,228 @@ +/** + * Contract: the OpenAI wire format converts to this engine's LLM schema without + * inventing, dropping, or silently reinterpreting anything a caller sent. + * + * These are the conversions that carry the real risk on POST /v1/chat/completions, + * because the format is fixed by clients that already exist and a mistranslation + * is invisible: the request succeeds and the model is simply given something else + * than the caller wrote. + */ +import { describe, expect, it } from "bun:test" +import type { ChatCompletionRequest } from "@redrob-code/protocol/groups/chat-completion" +import { + ConversionError, + toFinishReason, + toGeneration, + toLLMMessages, + toLLMTools, + toWireToolCalls, +} from "@redrob-code/server/handlers/chat-completion-wire" + +const base = { model: "redrob/auto" } as const + +function request(fields: Partial): ChatCompletionRequest { + return { ...base, messages: [], ...fields } as ChatCompletionRequest +} + +describe("messages", () => { + it("carries a plain exchange through in order", () => { + const messages = toLLMMessages( + request({ + messages: [ + { role: "system", content: "be brief" }, + { role: "user", content: "hello" }, + { role: "assistant", content: "hi" }, + ], + }), + ) + expect(messages.map((m) => m.role)).toEqual(["system", "user", "assistant"]) + expect(messages[1]!.content).toEqual([{ type: "text", text: "hello" }]) + }) + + it("treats developer as system, because OpenAI renamed the role and kept both", () => { + const messages = toLLMMessages(request({ messages: [{ role: "developer", content: "rules" }] })) + expect(messages).toHaveLength(1) + expect(messages[0]!.role).toBe("system") + }) + + it("drops an empty message instead of sending a blank turn", () => { + expect(toLLMMessages(request({ messages: [{ role: "system", content: "" }] }))).toHaveLength(0) + expect(toLLMMessages(request({ messages: [{ role: "assistant", content: null }] }))).toHaveLength(0) + }) + + it("parses tool_call arguments, which arrive as a JSON string and must not stay one", () => { + const messages = toLLMMessages( + request({ + messages: [ + { + role: "assistant", + tool_calls: [ + { id: "call_1", type: "function", function: { name: "read", arguments: '{"path":"a.txt"}' } }, + ], + }, + ], + }), + ) + const part = messages[0]!.content[0] as { type: string; id: string; name: string; input: unknown } + expect(part.type).toBe("tool-call") + expect(part.id).toBe("call_1") + expect(part.input).toEqual({ path: "a.txt" }) + }) + + it("keeps text and tool calls together on one assistant turn", () => { + const messages = toLLMMessages( + request({ + messages: [ + { + role: "assistant", + content: "let me look", + tool_calls: [{ id: "c1", type: "function", function: { name: "read", arguments: "{}" } }], + }, + ], + }), + ) + expect(messages[0]!.content.map((part) => part.type)).toEqual(["text", "tool-call"]) + }) + + it("refuses malformed tool_call arguments rather than forwarding an empty object", () => { + // Silently sending {} would make the model answer a question nobody asked. + expect(() => + toLLMMessages( + request({ + messages: [ + { + role: "assistant", + tool_calls: [{ id: "c1", type: "function", function: { name: "read", arguments: "{not json" } }], + }, + ], + }), + ), + ).toThrow(ConversionError) + }) + + it("refuses a tool message with no tool_call_id", () => { + // The provider pairs the result to the call by this id; without it the model + // sees an orphan result and there is nothing sensible to guess. + expect(() => toLLMMessages(request({ messages: [{ role: "tool", content: "done" }] }))).toThrow( + /requires tool_call_id/, + ) + }) + + it("carries a tool result under the id it answers", () => { + const messages = toLLMMessages( + request({ messages: [{ role: "tool", tool_call_id: "c1", name: "read", content: "file body" }] }), + ) + const part = messages[0]!.content[0] as { type: string; id: string } + expect(messages[0]!.role).toBe("tool") + expect(part.type).toBe("tool-result") + expect(part.id).toBe("c1") + }) + + it("accepts an array content part of type text", () => { + const messages = toLLMMessages( + request({ messages: [{ role: "user", content: [{ type: "text", text: "from a part" }] }] }), + ) + expect(messages[0]!.content).toEqual([{ type: "text", text: "from a part" }]) + }) + + it("refuses a content part it cannot carry instead of dropping it", () => { + // An image silently discarded looks to the user like the model ignored it. + expect(() => + toLLMMessages( + request({ + messages: [{ role: "user", content: [{ type: "image_url", image_url: { url: "data:..." } }] }], + }), + ), + ).toThrow(/not supported/) + }) +}) + +describe("tools", () => { + it("is undefined when the caller declared none, which is what selects engine-owned tools", () => { + expect(toLLMTools(request({}))).toBeUndefined() + expect(toLLMTools(request({ tools: [] }))).toBeUndefined() + }) + + it("converts a declaration and passes parameters through untouched", () => { + const parameters = { type: "object", properties: { path: { type: "string" } }, required: ["path"] } + const tools = toLLMTools( + request({ tools: [{ type: "function", function: { name: "read", description: "read a file", parameters } }] }), + ) + expect(tools!.definitions).toHaveLength(1) + expect(tools!.definitions[0]!.name).toBe("read") + expect(tools!.definitions[0]!.inputSchema).toEqual(parameters) + }) + + it("gives a parameterless tool an empty object schema rather than dropping it", () => { + const tools = toLLMTools(request({ tools: [{ type: "function", function: { name: "now" } }] })) + expect(tools!.definitions[0]!.inputSchema).toEqual({ type: "object", properties: {} }) + // An invented description would put words in the prompt the caller never wrote. + expect(tools!.definitions[0]!.description).toBe("") + }) + + it("carries tool_choice in both spellings", () => { + expect(toLLMTools(request({ tools: [{ type: "function", function: { name: "a" } }], tool_choice: "none" }))!.choice).toBe( + "none", + ) + expect( + toLLMTools( + request({ + tools: [{ type: "function", function: { name: "a" } }], + tool_choice: { type: "function", function: { name: "a" } }, + }), + )!.choice, + ).toEqual({ name: "a" }) + }) +}) + +describe("finish reason", () => { + it("maps the engine's vocabulary onto OpenAI's", () => { + expect(toFinishReason("stop")).toBe("stop") + expect(toFinishReason("length")).toBe("length") + expect(toFinishReason("tool-calls")).toBe("tool_calls") + expect(toFinishReason("content-filter")).toBe("content_filter") + expect(toFinishReason(undefined)).toBeNull() + }) + + it("does not report a failed turn as a clean stop", () => { + // "error" and "unknown" have no OpenAI equivalent. Reporting "stop" would + // tell the caller the turn ended normally when it did not. + expect(toFinishReason("error")).not.toBe("stop") + expect(toFinishReason("unknown")).not.toBe("stop") + }) +}) + +describe("tool calls out", () => { + it("re-stringifies input, because the engine parses it and clients expect a string", () => { + const wire = toWireToolCalls([{ id: "c1", name: "read", input: { path: "a.txt" } }]) + expect(wire[0]!.function.arguments).toBe('{"path":"a.txt"}') + expect(JSON.parse(wire[0]!.function.arguments)).toEqual({ path: "a.txt" }) + expect(wire[0]!.type).toBe("function") + }) + + it("emits {} for a call with no input, never undefined", () => { + // `arguments: undefined` breaks a client that calls JSON.parse on it. + expect(toWireToolCalls([{ id: "c1", name: "now", input: undefined }])[0]!.function.arguments).toBe("{}") + }) +}) + +describe("generation options", () => { + it("is undefined when the caller set none", () => { + expect(toGeneration(request({}))).toBeUndefined() + }) + + it("renames the wire fields to the engine's", () => { + const generation = toGeneration(request({ max_tokens: 256, temperature: 0.2, top_p: 0.9 })) + expect(generation).toEqual({ maxTokens: 256, temperature: 0.2, topP: 0.9 }) + }) + + it("lets max_tokens win over max_completion_tokens when both are sent", () => { + expect(toGeneration(request({ max_tokens: 100, max_completion_tokens: 900 }))!["maxTokens"]).toBe(100) + expect(toGeneration(request({ max_completion_tokens: 900 }))!["maxTokens"]).toBe(900) + }) + + it("normalizes a single stop string to a list", () => { + expect(toGeneration(request({ stop: "END" }))!["stop"]).toEqual(["END"]) + expect(toGeneration(request({ stop: ["A", "B"] }))!["stop"]).toEqual(["A", "B"]) + }) +}) diff --git a/packages/server/src/handlers/chat-completion-wire.ts b/packages/server/src/handlers/chat-completion-wire.ts new file mode 100644 index 0000000000..636bd0c564 --- /dev/null +++ b/packages/server/src/handlers/chat-completion-wire.ts @@ -0,0 +1,207 @@ +/** + * Translation between the OpenAI chat-completions wire format and this engine's + * LLM schema. Pure: no Effect, no services, no I/O — so the parts most likely to + * be wrong are the parts that can be tested directly. + * + * Direction matters here. Requests come in from a caller we do not control, so + * every field is validated or ignored rather than trusted. Responses go out to + * clients that already exist (redrob-browser builds and parses this format + * today), so the output shape is fixed by them, not by convenience. + */ +import { Message, ToolCallPart, ToolDefinition, ToolResultPart } from "@redrob-code/llm" +import type { ChatCompletionRequest } from "@redrob-code/protocol/groups/chat-completion" + +/** What the caller declared, after validation. */ +export interface ToolsIn { + readonly definitions: ReadonlyArray + readonly choice: "auto" | "none" | "required" | { readonly name: string } | undefined +} + +export class ConversionError extends Error { + constructor( + message: string, + readonly param: string | undefined, + ) { + super(message) + this.name = "ConversionError" + } +} + +/** + * OpenAI allows string content or an array of parts. The array form is passed + * through as text parts only: media parts need a mediaType this route has no + * validated source for, and silently dropping an image would be worse than + * saying so. + */ +function contentParts( + content: ChatCompletionRequest["messages"][number]["content"], + where: string, +): ReadonlyArray> { + if (content === undefined || content === null) return [] + if (typeof content === "string") return content.length > 0 ? [Message.text(content)] : [] + const parts: Array> = [] + for (const part of content) { + const type = part["type"] + if (type === "text" && typeof part["text"] === "string") { + parts.push(Message.text(part["text"] as string)) + continue + } + throw new ConversionError( + `${where}: content part of type ${String(type)} is not supported on this route`, + "messages", + ) + } + return parts +} + +/** + * `system` and `developer` both become a system message: OpenAI renamed the role + * and kept accepting the old one, so a caller may send either. + * + * A `tool` message must name the call it answers. Without `tool_call_id` the + * provider cannot pair it with the assistant turn that asked, and the model sees + * an orphan result — so this is rejected rather than guessed at. + */ +export function toLLMMessages(request: ChatCompletionRequest): ReadonlyArray { + const out: Array = [] + request.messages.forEach((message, index) => { + const where = `messages[${index}]` + switch (message.role) { + case "system": + case "developer": { + const text = typeof message.content === "string" ? message.content : "" + if (text.length > 0) out.push(Message.system(text)) + return + } + case "user": { + out.push(Message.make({ role: "user", content: [...contentParts(message.content, where)] })) + return + } + case "assistant": { + // An assistant turn being replayed can carry text, tool calls, or both. + const parts: Array | ToolCallPart> = [ + ...contentParts(message.content, where), + ] + for (const call of message.tool_calls ?? []) { + parts.push( + ToolCallPart.make({ + id: call.id, + name: call.function.name, + // The wire carries a JSON string; the engine wants the parsed value. + // A malformed string is the caller's bug and must not be forwarded + // as a silently empty argument object. + input: parseArguments(call.function.arguments, `${where}.tool_calls`), + }), + ) + } + if (parts.length > 0) out.push(Message.make({ role: "assistant", content: parts })) + return + } + case "tool": { + if (message.tool_call_id === undefined || message.tool_call_id.length === 0) { + throw new ConversionError(`${where}: a tool message requires tool_call_id`, "messages") + } + out.push( + Message.tool( + ToolResultPart.make({ + id: message.tool_call_id, + name: message.name ?? "", + result: typeof message.content === "string" ? message.content : "", + }), + ), + ) + return + } + } + }) + return out +} + +function parseArguments(raw: string, where: string): unknown { + if (raw.length === 0) return {} + try { + return JSON.parse(raw) + } catch { + throw new ConversionError(`${where}: arguments is not valid JSON`, "messages") + } +} + +/** + * A declared tool with no `parameters` still needs a schema the provider will + * accept, so it becomes an empty object schema rather than being dropped. + * + * `description` is required by ToolDefinition and optional on the wire; an empty + * string is the honest default — inventing one would put words in the model's + * prompt that the caller never wrote. + */ +export function toLLMTools(request: ChatCompletionRequest): ToolsIn | undefined { + if (request.tools === undefined || request.tools.length === 0) return undefined + const definitions = request.tools.map((tool, index) => { + if (tool.function.name.length === 0) { + throw new ConversionError(`tools[${index}]: function.name is required`, "tools") + } + return new ToolDefinition({ + name: tool.function.name, + description: tool.function.description ?? "", + inputSchema: tool.function.parameters ?? { type: "object", properties: {} }, + }) + }) + return { definitions, choice: toToolChoice(request.tool_choice) } +} + +function toToolChoice(choice: ChatCompletionRequest["tool_choice"]): ToolsIn["choice"] { + if (choice === undefined) return undefined + if (typeof choice === "string") return choice + return { name: choice.function.name } +} + +/** OpenAI's finish_reason vocabulary. The engine's is close but not identical. */ +export function toFinishReason(reason: string | undefined): string | null { + switch (reason) { + case "stop": + return "stop" + case "length": + return "length" + case "tool-calls": + return "tool_calls" + case "content-filter": + return "content_filter" + case undefined: + return null + default: + // "error" and "unknown" have no OpenAI equivalent. Reporting them as + // "stop" would tell the caller the turn ended cleanly when it did not. + return reason + } +} + +/** The engine parses tool-call input; OpenAI clients expect a JSON string. */ +export function toWireToolCalls( + calls: ReadonlyArray<{ readonly id: string; readonly name: string; readonly input: unknown }>, +): ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } +}> { + return calls.map((call) => ({ + id: call.id, + type: "function" as const, + function: { name: call.name, arguments: JSON.stringify(call.input ?? {}) }, + })) +} + +/** `max_tokens` wins over `max_completion_tokens` when a caller sends both. */ +export function toGeneration(request: ChatCompletionRequest): Record | undefined { + const generation: Record = {} + const maxTokens = request.max_tokens ?? request.max_completion_tokens + if (maxTokens !== undefined) generation["maxTokens"] = maxTokens + if (request.temperature !== undefined) generation["temperature"] = request.temperature + if (request.top_p !== undefined) generation["topP"] = request.top_p + if (request.seed !== undefined) generation["seed"] = request.seed + if (request.frequency_penalty !== undefined) generation["frequencyPenalty"] = request.frequency_penalty + if (request.presence_penalty !== undefined) generation["presencePenalty"] = request.presence_penalty + if (request.stop !== undefined) { + generation["stop"] = typeof request.stop === "string" ? [request.stop] : [...request.stop] + } + return Object.keys(generation).length > 0 ? generation : undefined +} From 306adad4c8d0dd257582b51a8458ce10dea24e59 Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Tue, 22 Sep 2026 06:29:16 +0000 Subject: [PATCH 2/7] feat(api): serve POST /v1/chat/completions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The handler for the contract in the previous commit. The route is now reachable. It touches no Session, Agent or tool runtime: it resolves a provider the way the agent does (catalog entry plus the active integration credential), makes one model call, and translates. Everything stateful stays in the session API. handleRaw rather than handle, because the reply is JSON or text/event-stream depending on `stream`, and only the raw form chooses its own response. The cost is that the payload is not decoded for you, so the declared schema is applied in the handler rather than the body being trusted. Streaming is where the care went. Tool calls are emitted from the completed `tool-call` event, not from `tool-input-delta`: the engine parses arguments and the wire wants a JSON string, so re-emitting fragments would mean re-serializing a half-parsed value. And `provider-error` arrives as a stream ELEMENT rather than a failure — left unhandled, an upstream 500 would end the stream with a clean [DONE] and the caller would read a truncated answer as a complete one, so it is matched explicitly and sent as an error frame. Once headers are out, a frame is the only way to say the turn broke. Error mapping keeps one value load-bearing: 401 with code engine_not_authenticated, covering both a missing credential and a rejected one — from the caller's side those are one problem and neither is fixed by retrying. What is NOT remapped is a content-policy classification: the executor checks the body before the status, so an upstream 401 mentioning safety arrives as ContentPolicy, and reclassifying it here would report a policy refusal as a login problem. A caller-supplied base URL is deliberately not accepted. The config layer only admits a new openai-compatible provider at a local address (isLocalEndpoint); honouring an arbitrary URL here would bypass that and forward the engine's own credential to whatever host the caller named. A local model is selected by its `provider/model` id instead. Four APIs were guessed wrong first and corrected against the source, not worked around: this Effect version has no Effect.either, no Effect.catchAll and no Stream.mapConcat, LLMResponse carries finishReason rather than a finish object, and catalog.model.get returns an Effect on the Service (only the Draft form is synchronous). tool_choice now travels as the engine's own shape instead of a bare string. A bare string is ambiguous — the normalizer reads auto/none/required as modes and anything else as a tool name — so a tool actually named `auto` would silently become a mode. A test pins that case. 23 tests pass; protocol and server typecheck clean. Still to come: the capabilities flag, the httpapi-exercise scenario (test:httpapi runs --fail-on-missing, so CI will fail until it exists), the SSE OpenAPI patch in public.ts, and a route test. --- packages/protocol/src/api.ts | 5 + .../test/server/chat-completion-wire.test.ts | 24 +- packages/server/src/handlers.ts | 2 + .../src/handlers/chat-completion-wire.ts | 48 ++- .../server/src/handlers/chat-completion.ts | 292 ++++++++++++++++++ 5 files changed, 354 insertions(+), 17 deletions(-) create mode 100644 packages/server/src/handlers/chat-completion.ts diff --git a/packages/protocol/src/api.ts b/packages/protocol/src/api.ts index e7256237ca..9280a31083 100644 --- a/packages/protocol/src/api.ts +++ b/packages/protocol/src/api.ts @@ -22,6 +22,7 @@ import { LocationGroup } from "./groups/location" import { IntegrationGroup } from "./groups/integration" import { CredentialGroup } from "./groups/credential" import { ProjectCopyGroup } from "./groups/project-copy" +import { ChatCompletionGroup } from "./groups/chat-completion" // Protocol owns middleware placement, while Server injects concrete keys so Core service identities stay downstream. const makeApiFromGroup = < @@ -55,6 +56,10 @@ const makeApiFromGroup = < .add(makeQuestionGroup(locationMiddleware, sessionLocationMiddleware)) .add(ReferenceGroup.middleware(locationMiddleware)) .add(ProjectCopyGroup.middleware(locationMiddleware)) + // The OpenAI-compatible inference route. It takes locationMiddleware like the + // rest: the provider registry it resolves against is per-location, so a + // project-local provider must be visible to it. + .add(ChatCompletionGroup.middleware(locationMiddleware)) .annotateMerge( OpenApi.annotations({ title: "redrob HttpApi", diff --git a/packages/redrob/test/server/chat-completion-wire.test.ts b/packages/redrob/test/server/chat-completion-wire.test.ts index d3bf755954..b2dde1ffd6 100644 --- a/packages/redrob/test/server/chat-completion-wire.test.ts +++ b/packages/redrob/test/server/chat-completion-wire.test.ts @@ -160,10 +160,13 @@ describe("tools", () => { expect(tools!.definitions[0]!.description).toBe("") }) - it("carries tool_choice in both spellings", () => { - expect(toLLMTools(request({ tools: [{ type: "function", function: { name: "a" } }], tool_choice: "none" }))!.choice).toBe( - "none", - ) + it("carries tool_choice as the engine's shape, never as a bare string", () => { + // A bare string is ambiguous: the engine's normalizer reads + // "auto"/"none"/"required" as modes and anything else as a tool name, so a + // tool actually named `auto` would silently become a mode. + expect( + toLLMTools(request({ tools: [{ type: "function", function: { name: "a" } }], tool_choice: "none" }))!.choice, + ).toEqual({ type: "none" }) expect( toLLMTools( request({ @@ -171,7 +174,18 @@ describe("tools", () => { tool_choice: { type: "function", function: { name: "a" } }, }), )!.choice, - ).toEqual({ name: "a" }) + ).toEqual({ type: "tool", name: "a" }) + }) + + it("names a tool called auto as a tool, not as the auto mode", () => { + expect( + toLLMTools( + request({ + tools: [{ type: "function", function: { name: "auto" } }], + tool_choice: { type: "function", function: { name: "auto" } }, + }), + )!.choice, + ).toEqual({ type: "tool", name: "auto" }) }) }) diff --git a/packages/server/src/handlers.ts b/packages/server/src/handlers.ts index 2f7bb9e92d..ab62df08e9 100644 --- a/packages/server/src/handlers.ts +++ b/packages/server/src/handlers.ts @@ -18,6 +18,7 @@ import { LocationHandler } from "./handlers/location" import { IntegrationHandler } from "./handlers/integration" import { CredentialHandler } from "./handlers/credential" import { ProjectCopyHandler } from "./handlers/project-copy" +import { ChatCompletionHandler } from "./handlers/chat-completion" export const handlers = Layer.mergeAll( HealthHandler, @@ -39,4 +40,5 @@ export const handlers = Layer.mergeAll( QuestionHandler, ReferenceHandler, ProjectCopyHandler, + ChatCompletionHandler, ) diff --git a/packages/server/src/handlers/chat-completion-wire.ts b/packages/server/src/handlers/chat-completion-wire.ts index 636bd0c564..40dd82fdb6 100644 --- a/packages/server/src/handlers/chat-completion-wire.ts +++ b/packages/server/src/handlers/chat-completion-wire.ts @@ -14,7 +14,12 @@ import type { ChatCompletionRequest } from "@redrob-code/protocol/groups/chat-co /** What the caller declared, after validation. */ export interface ToolsIn { readonly definitions: ReadonlyArray - readonly choice: "auto" | "none" | "required" | { readonly name: string } | undefined + /** + * The engine's own shape, not the wire's. A bare string would be ambiguous: + * its normalizer reads "auto"/"none"/"required" as modes and anything else as + * a tool name, so a tool actually named `auto` would silently become a mode. + */ + readonly choice: { readonly type: "auto" | "none" | "required" | "tool"; readonly name?: string } | undefined } export class ConversionError extends Error { @@ -151,8 +156,8 @@ export function toLLMTools(request: ChatCompletionRequest): ToolsIn | undefined function toToolChoice(choice: ChatCompletionRequest["tool_choice"]): ToolsIn["choice"] { if (choice === undefined) return undefined - if (typeof choice === "string") return choice - return { name: choice.function.name } + if (typeof choice === "string") return { type: choice } + return { type: "tool", name: choice.function.name } } /** OpenAI's finish_reason vocabulary. The engine's is close but not identical. */ @@ -191,17 +196,36 @@ export function toWireToolCalls( } /** `max_tokens` wins over `max_completion_tokens` when a caller sends both. */ -export function toGeneration(request: ChatCompletionRequest): Record | undefined { - const generation: Record = {} +export function toGeneration(request: ChatCompletionRequest): GenerationInput | undefined { + const generation: { + maxTokens?: number + temperature?: number + topP?: number + seed?: number + frequencyPenalty?: number + presencePenalty?: number + stop?: ReadonlyArray + } = {} const maxTokens = request.max_tokens ?? request.max_completion_tokens - if (maxTokens !== undefined) generation["maxTokens"] = maxTokens - if (request.temperature !== undefined) generation["temperature"] = request.temperature - if (request.top_p !== undefined) generation["topP"] = request.top_p - if (request.seed !== undefined) generation["seed"] = request.seed - if (request.frequency_penalty !== undefined) generation["frequencyPenalty"] = request.frequency_penalty - if (request.presence_penalty !== undefined) generation["presencePenalty"] = request.presence_penalty + if (maxTokens !== undefined) generation.maxTokens = maxTokens + if (request.temperature !== undefined) generation.temperature = request.temperature + if (request.top_p !== undefined) generation.topP = request.top_p + if (request.seed !== undefined) generation.seed = request.seed + if (request.frequency_penalty !== undefined) generation.frequencyPenalty = request.frequency_penalty + if (request.presence_penalty !== undefined) generation.presencePenalty = request.presence_penalty if (request.stop !== undefined) { - generation["stop"] = typeof request.stop === "string" ? [request.stop] : [...request.stop] + generation.stop = typeof request.stop === "string" ? [request.stop] : [...request.stop] } return Object.keys(generation).length > 0 ? generation : undefined } + +/** The subset of the engine's generation options this route can be asked for. */ +export interface GenerationInput { + readonly maxTokens?: number + readonly temperature?: number + readonly topP?: number + readonly seed?: number + readonly frequencyPenalty?: number + readonly presencePenalty?: number + readonly stop?: ReadonlyArray +} diff --git a/packages/server/src/handlers/chat-completion.ts b/packages/server/src/handlers/chat-completion.ts new file mode 100644 index 0000000000..dee9c6a36b --- /dev/null +++ b/packages/server/src/handlers/chat-completion.ts @@ -0,0 +1,292 @@ +/** + * `POST /v1/chat/completions` — see docs/LOCAL-ENGINE-API.md and + * protocol/groups/chat-completion.ts for the contract. + * + * This handler deliberately does NOT touch Session, Agent or the tool runtime. + * It resolves a provider the way the agent does, makes one model call, and + * translates. Everything stateful belongs to the session API. + * + * `handleRaw` rather than `handle`, because the reply is JSON or + * text/event-stream depending on `stream`, and only the raw form can return an + * HttpServerResponse of its own choosing. + */ +import { Catalog } from "@redrob-code/core/catalog" +import { Integration } from "@redrob-code/core/integration" +import { ModelV2 } from "@redrob-code/core/model" +import { ProviderV2 } from "@redrob-code/core/provider" +import { fromCatalogModel } from "@redrob-code/core/session/runner/model" +import { LLM, LLMError } from "@redrob-code/llm" +import type { LLMEvent, LLMResponse } from "@redrob-code/llm" +import type { Credential } from "@redrob-code/schema/credential" +import { ChatCompletionRequest } from "@redrob-code/protocol/groups/chat-completion" +import { Effect, Stream } from "effect" +import { HttpServerRequest, HttpServerResponse } from "effect/unstable/http" +import { HttpApiBuilder } from "effect/unstable/httpapi" +import * as Sse from "effect/unstable/encoding/Sse" +import { Api } from "../api" +import { + ConversionError, + toFinishReason, + toGeneration, + toLLMMessages, + toLLMTools, + toWireToolCalls, +} from "./chat-completion-wire" + +/** OpenAI ids are `chatcmpl-`; clients log and correlate on them. */ +function completionID(): string { + return `chatcmpl-${crypto.randomUUID().replaceAll("-", "")}` +} + +function seconds(): number { + return Math.floor(Date.now() / 1000) +} + +interface WireError { + readonly status: 400 | 401 | 429 | 502 + readonly message: string + readonly type: string + readonly code: string | null +} + +/** + * Maps an engine failure onto the OpenAI error envelope. + * + * `engine_not_authenticated` is the load-bearing value: it is the only code a + * client may turn into a sign-in prompt, and it covers both a missing credential + * and one the provider rejected — from the caller's side those are the same + * problem, and neither is fixed by retrying. + * + * Note what is NOT mapped to auth. The executor classifies a content-policy body + * BEFORE it looks at the status, so an upstream 401 whose body mentions safety + * arrives as ContentPolicy. That ordering is upstream's and this reads whatever + * it produced rather than second-guessing it — reclassifying here would report a + * policy refusal as a login problem. + */ +function toWireError(error: unknown): WireError { + if (error instanceof ConversionError) { + return { status: 400, message: error.message, type: "invalid_request_error", code: null } + } + if (error instanceof LLMError) { + const reason = error.reason + switch (reason._tag) { + case "Authentication": + return { + status: 401, + message: reason.message, + type: "authentication_error", + code: "engine_not_authenticated", + } + case "InvalidRequest": + return { status: 400, message: reason.message, type: "invalid_request_error", code: null } + case "RateLimit": + return { status: 429, message: reason.message, type: "rate_limit_error", code: "rate_limit_exceeded" } + case "QuotaExceeded": + return { status: 429, message: reason.message, type: "insufficient_quota", code: "insufficient_quota" } + case "ContentPolicy": + return { status: 400, message: reason.message, type: "invalid_request_error", code: "content_policy_violation" } + default: + return { status: 502, message: reason.message, type: "api_error", code: null } + } + } + const message = error instanceof Error ? error.message : String(error) + return { status: 502, message, type: "api_error", code: null } +} + +function errorResponse(error: unknown): Effect.Effect { + const wire = toWireError(error) + return HttpServerResponse.json( + { error: { message: wire.message, type: wire.type, code: wire.code, param: null } }, + { status: wire.status }, + ).pipe(Effect.orDie) +} + +/** + * Resolves the requested model the way the agent does: catalog entry plus the + * active integration credential. `redrob/auto` is the Console route; any other + * `provider/model` string resolves against the same registry, which is how a + * locally served model becomes reachable without a second code path. + * + * A caller-supplied base URL is deliberately NOT accepted. The config layer only + * admits a new openai-compatible provider at a local address (isLocalEndpoint); + * honouring an arbitrary URL here would bypass that check and forward the + * engine's own credential to whatever host the caller named. + */ +const resolveModel = Effect.fn("chat.resolveModel")(function* (model: string) { + const [providerPart, ...rest] = model.split("/") + const providerID = ProviderV2.ID.make(rest.length > 0 ? providerPart! : "redrob") + const modelID = ModelV2.ID.make(rest.length > 0 ? rest.join("/") : model) + + const catalog = yield* Catalog.Service + const info = yield* catalog.model.get(providerID, modelID) + if (info === undefined) { + return yield* Effect.fail( + new ConversionError(`model ${model} is not available on this engine`, "model"), + ) + } + + const integrations = yield* Integration.Service + let credential: Credential.Value | undefined + // `active` has no error channel; only `resolve` can fail, and a failure there + // means "no usable credential", which the provider layer reports better than a + // 500 from here would. + const connection = yield* integrations.connection.active(Integration.ID.make(providerID)) + if (connection !== undefined) { + credential = yield* integrations.connection + .resolve(connection) + .pipe(Effect.catch(() => Effect.succeed(undefined))) + } + return yield* fromCatalogModel(info, credential) +}) + +const buildRequest = Effect.fn("chat.buildRequest")(function* (payload: ChatCompletionRequest) { + const model = yield* resolveModel(payload.model) + const messages = toLLMMessages(payload) + const tools = toLLMTools(payload) + const generation = toGeneration(payload) + return LLM.request({ + model, + messages, + // Absent tools is what selects engine-owned behaviour; an empty array would + // tell the provider "you may call nothing", which is a different thing. + ...(tools ? { tools: tools.definitions, ...(tools.choice ? { toolChoice: tools.choice } : {}) } : {}), + ...(generation ? { generation } : {}), + }) +}) + +function toJsonBody(id: string, model: string, response: LLMResponse): unknown { + const toolCalls = toWireToolCalls( + response.toolCalls.map((call) => ({ id: call.id, name: call.name, input: call.input })), + ) + const usage = response.usage + return { + id, + object: "chat.completion", + created: seconds(), + model, + choices: [ + { + index: 0, + message: { + role: "assistant", + content: response.text.length > 0 ? response.text : null, + ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}), + }, + finish_reason: toFinishReason(response.finishReason), + }, + ], + ...(usage + ? { + usage: { + prompt_tokens: usage.inputTokens ?? 0, + completion_tokens: usage.outputTokens ?? 0, + total_tokens: usage.totalTokens ?? 0, + }, + } + : {}), + } +} + +/** One SSE frame carrying an OpenAI chunk. */ +function chunkFrame(id: string, model: string, delta: unknown, finishReason: string | null): Sse.Event { + return { + _tag: "Event", + event: "message", + id: undefined, + data: JSON.stringify({ + id, + object: "chat.completion.chunk", + created: seconds(), + model, + choices: [{ index: 0, delta, finish_reason: finishReason }], + }), + } +} + +function doneFrame(): Sse.Event { + return { _tag: "Event", event: "message", id: undefined, data: "[DONE]" } +} + +/** + * Translates the engine's event stream into OpenAI chunks. + * + * Two things here are not obvious. Tool-call fragments are emitted from the + * COMPLETED `tool-call` event rather than from `tool-input-delta`, because the + * engine parses arguments and the wire wants a JSON string — re-emitting + * fragments would mean re-serializing a half-parsed value. And `provider-error` + * arrives as a stream ELEMENT, not a failure, so it is matched explicitly: left + * unhandled, an upstream 500 would end the stream with a clean `[DONE]` and the + * caller would read a truncated answer as a complete one. + */ +function toChunks(id: string, model: string, events: Stream.Stream) { + return events.pipe( + Stream.flatMap((event: LLMEvent) => Stream.fromIterable(eventFrames(id, model, event))), + Stream.concat(Stream.make(doneFrame())), + ) +} + +function eventFrames(id: string, model: string, event: LLMEvent): ReadonlyArray { + switch (event.type) { + case "text-delta": + return [chunkFrame(id, model, { content: event.text }, null)] + case "reasoning-delta": + return [chunkFrame(id, model, { reasoning_content: event.text }, null)] + case "tool-call": { + const [call] = toWireToolCalls([{ id: event.id, name: event.name, input: event.input }]) + return [chunkFrame(id, model, { tool_calls: [{ index: 0, ...call }] }, null)] + } + case "finish": + return [chunkFrame(id, model, {}, toFinishReason(event.reason))] + case "provider-error": + // Surfaced as data, not as a stream failure: headers are already sent, so + // this is the only way the caller learns the turn broke. + return [ + { + _tag: "Event" as const, + event: "message", + id: undefined, + data: JSON.stringify({ + error: { message: event.message, type: "api_error", code: null, param: null }, + }), + }, + ] + default: + return [] + } +} + +export const ChatCompletionHandler = HttpApiBuilder.group(Api, "server.chat", (handlers) => + Effect.gen(function* () { + return handlers.handleRaw("chat.completions", () => + Effect.gen(function* () { + // handleRaw does not decode the payload — that is the cost of being able + // to answer with either JSON or an event stream. The declared schema is + // still the contract, so it is applied here rather than trusting the body. + const payload = yield* HttpServerRequest.schemaBodyJson(ChatCompletionRequest) + const id = completionID() + const request = yield* buildRequest(payload) + + if (payload.stream !== true) { + const response = yield* LLM.generate(request) + return yield* HttpServerResponse.json(toJsonBody(id, payload.model, response)).pipe(Effect.orDie) + } + + // Streaming: the error envelope can only be sent before the first byte, + // so a failure after headers travels as an SSE error frame (see + // eventFrames) rather than as a status code. + const frames = toChunks(id, payload.model, LLM.stream(request)).pipe( + Stream.pipeThroughChannel(Sse.encode()), + Stream.encodeText, + ) + return HttpServerResponse.stream(frames, { + contentType: "text/event-stream", + headers: { + "Cache-Control": "no-cache, no-transform", + "X-Accel-Buffering": "no", + "X-Content-Type-Options": "nosniff", + }, + }) + }).pipe(Effect.catch((error) => errorResponse(error))), + ) + }), +) From e22989174d05a9275bf685c1bd4500134b7ff30e Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Tue, 22 Sep 2026 06:44:28 +0000 Subject: [PATCH 3/7] docs: provider authentication, and what each vendor actually permits The design of record for how a user connects their own model access, so one setup serves code, cowork, office and whatever comes after. The product ask was 'let people use the Claude or ChatGPT they already pay for'. Half of that is not available, and the split is contractual rather than technical, so the document leads with the survey and links every claim. Not available: a sign-in button in our own apps that spends a consumer subscription. Anthropic's compliance page forbids third parties offering Claude.ai login or routing through Free/Pro/Max credentials, and bans accounts for spoofing the Claude Code harness - our own upstream removed Claude support on 2026-02-19 under legal request. Google's Gemini CLI FAQ says piggybacking its OAuth may terminate the account, and there is a documented mass-ban thread. OpenAI allows it in practice, but only by embedding OpenAI's own Codex runtime, and no clause grants it, so it is a revocable product decision. Available, and it covers most of the intent: BYOK everywhere (Anthropic and Google both name a user API key as THE supported third-party path), plus three real sign-in buttons - GitHub Copilot through the Copilot SDK, which spends the user's own Copilot subscription with GitHub's blessing and a documented desktop device flow; OpenRouter OAuth PKCE, which reaches Claude and GPT models behind an account login and is the closest legitimate thing to what was asked for; and Azure OpenAI via Entra device code. The design itself adds little and removes a lot. Credentials stay in this engine's existing auth.json, and products stop having stores of their own - today Office, Design, the extension, Query and Recall keep five stores that never read each other, which is the whole reason a user logs in again in every app. Products never hold a key: /v1/chat/completions already carries the engine's credential outward and takes none from the caller, so adding a provider is one change here rather than a release in every app, and an app that holds no key cannot leak one. Four routes are proposed to expose the login flow that already exists in , with three rules: the key never comes back out, the OAuth code stays in the engine, and no caller-supplied base URL - that last one because honouring it would forward this engine's credential to whatever host the caller named. Also records a compliance check to keep: this repository contains no Anthropic OAuth client id and no claude.ai endpoint, and the only hardcoded vendor OAuth is xAI's. --- docs/PROVIDER-AUTH.md | 189 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 189 insertions(+) create mode 100644 docs/PROVIDER-AUTH.md diff --git a/docs/PROVIDER-AUTH.md b/docs/PROVIDER-AUTH.md new file mode 100644 index 0000000000..e478ac6413 --- /dev/null +++ b/docs/PROVIDER-AUTH.md @@ -0,0 +1,189 @@ +# Provider authentication + +How a user connects their own model access to this engine, and therefore to every +product built on it. One credential store, read by all of them, so a user sets up +once rather than once per app. + +The goal this answers is "let people use the Claude or ChatGPT access they already +pay for". Part of that is available and part of it is not, and the split is not +technical — it is what each vendor's terms permit a third-party application to do. +So the shapes are enumerated first, then the design. + +Not legal advice. Every claim below links the document it came from, checked +2026-09-22; re-check before shipping, because three of these pages changed in the +first half of this year. + +## What each vendor actually allows a third-party app + +| vendor | "sign in" for a third-party app | user's own API key | user's consumer subscription | +| --- | --- | --- | --- | +| Anthropic | **No** — expressly forbidden | Yes, expressly | **No** — prohibited, enforced with account bans | +| OpenAI | No self-serve registration; granted case by case | Yes | Only by embedding OpenAI's own Codex runtime | +| Google Gemini | OAuth exists but bills *your* Cloud project | Yes — the named supported path | **No** — prohibited, enforced with bans | +| GitHub Copilot | **Yes** — the Copilot SDK, officially | Yes | Yes, billed to the user's own subscription | +| OpenRouter | **Yes** — OAuth PKCE | Yes | n/a (it is BYOK by design) | +| Azure OpenAI | **Yes** — Microsoft Entra ID | Yes | n/a (user's own Azure resource) | +| Amazon Bedrock | No (SigV4 / Bedrock API keys) | Yes — the user's own AWS account | n/a | +| AWS Kiro | **No** | Kiro API key, subscriber-only | Only by driving the user's own installed CLI | + +Sources, and the sentences that decide it: + +- Anthropic, [Claude Code legal and compliance](https://code.claude.com/docs/en/legal-and-compliance): + "Anthropic does not permit third-party developers to offer Claude.ai login into + their own applications, or to route requests through Free, Pro, or Max plan + credentials on behalf of their users." And developers "should use API key + authentication through Claude Console or a supported cloud provider." Enforced: + accounts were banned for spoofing the Claude Code harness, and our own upstream + (OpenCode) removed Claude subscription AND Claude API key support on 2026-02-19 + citing Anthropic legal requests ([The Register, + 2026-02-20](https://www.theregister.com/software/2026/02/20/anthropic-clarifies-ban-on-third-party-tool-access-to-claude/5014546)). + The only sanctioned subscription route is shipping Claude Code itself, + unmodified, with its auth methods intact. +- OpenAI, [Codex authentication](https://developers.openai.com/codex/auth/): + "Sign in with ChatGPT" is documented for OpenAI's own surfaces only. Third + parties that have it — Zed, OpenClaw — get there by wrapping OpenAI's own Codex + runtime ([Zed](https://zed.dev/blog/chatgpt-subscription-in-zed)). Treat that as + a revocable product decision, not an entitlement: no terms clause grants it. +- Google, [Gemini CLI FAQ](https://github.com/google-gemini/gemini-cli/blob/main/docs/resources/faq.md): + "the supported and secure method is to use a Vertex AI or Google AI Studio API + key", and piggybacking Gemini CLI's OAuth "may be grounds for immediate + suspension or termination". Enforced in + [this thread](https://github.com/google-gemini/gemini-cli/discussions/20632). + Note the Gemini API terms also say the API is "not for consumer use" and + require paid services for EEA/UK/CH users. +- GitHub, [Copilot SDK OAuth setup](https://docs.github.com/en/copilot/how-tos/copilot-sdk/setup/github-oauth): + "Copilot requests are made on behalf of each authenticated user, using their + Copilot subscription… Your app never handles model API keys." A device flow is + documented for exactly our case, "Desktop applications where users interact + directly". +- OpenRouter, [OAuth PKCE](https://openrouter.ai/docs/guides/overview/auth/oauth): + send the user to `/auth` with a `code_challenge`, exchange the code for a + **user-controlled API key**. Loopback callback on any port, plus a headless + paste mode. +- Azure, [Entra ID auth](https://learn.microsoft.com/azure/ai-services/openai/how-to/managed-identity) + with the [device authorization grant](https://learn.microsoft.com/en-us/entra/identity-platform/v2-oauth2-device-code). +- Kiro, [Authentication](https://kiro.dev/docs/getting-started/authentication/): + subscriber `ksk_` API keys exist; the only documented embed path is driving the + user's own CLI over [ACP](https://kiro.dev/docs/cli/acp/). + +### What this means for the product ask + +"Sign in with Claude" and "Sign in with ChatGPT", as buttons in our own apps +spending the user's consumer subscription, are not available. Building them means +impersonating a first-party client, and the vendors ban accounts for it — the cost +lands on our users, not on us. + +What IS available, and covers most of the intent: + +1. **BYOK for every provider.** Anthropic and Google both name this as the + supported path for third-party tools. A user with a Claude API key or a Gemini + key connects in one step. +2. **Three real sign-in buttons**: GitHub Copilot, OpenRouter, Azure OpenAI. The + first two are the interesting ones — Copilot spends the user's own Copilot + subscription with GitHub's blessing, and OpenRouter fronts Claude and GPT + models behind an account login, which is the closest legitimate thing to what + was asked for. +3. **Optionally, unmodified first-party binaries.** Shipping Claude Code or the + Codex CLI as-is and letting it authenticate itself is sanctioned by both + vendors. It is a different product shape — their harness, not ours — so it is + recorded here as available rather than recommended. + +## Design + +### One store, in the engine + +Credentials live where they already live: `Global.Path.data/auth.json`, through +`packages/redrob/src/auth`, whose `Info` union is already `Oauth | Api | +WellKnown`. `packages/core/src/console-key.ts` reads env → credential store → +that file, and `Integration` resolves a connection into a `Credential.Value` for +the provider layer. + +Nothing new is invented, because the point is that products stop having stores of +their own. Today Office keeps keys in `userData/ai-settings.json`, Design in the +macOS keychain, the extension in `chrome.storage.local`, Query and Recall in +separate keyring services — five stores that never read each other, which is why +a user logs in again in every app. + +A product reads the engine's store instead. It keeps its own as a cache if it +wants, but the engine's file is the source of truth, and the engine is what makes +the call. + +### Products never hold a key + +`POST /v1/chat/completions` (see `docs/LOCAL-ENGINE-API.md`) already carries the +engine's credential outward and takes none from the caller. That is the whole +mechanism: a product names a `model`, the engine resolves the provider and its +credential. Adding a provider is then a change in one place, and every downstream +app gets it without shipping a release. + +This also removes a class of bug rather than moving it: a product that holds no +key cannot leak one, log one, or sync one. + +### Connecting a provider + +`redrob providers login` already implements both shapes — a generic OAuth flow +with `authorize()` plus `auto` and `code` callbacks, and an API-key path +(`packages/redrob/src/cli/cmd/providers.ts`). What is missing is not the flow but +its exposure: a product cannot drive it today. + +Add, on the v2 surface beside the completions route: + +- `GET /v1/providers` — what this engine can use, each entry declaring which auth + methods it accepts (`oauth`, `api-key`) and whether a credential is present. + This is what lets an app render a settings page without hardcoding a vendor + list that goes stale. +- `POST /v1/providers/:id/login` — starts a flow. For OAuth it returns the + authorization URL and an opaque attempt id; for an API key it accepts the key. +- `POST /v1/providers/:id/login/:attempt` — completes an OAuth attempt with the + authorization code, or reports that the loopback callback already completed it. +- `DELETE /v1/providers/:id/credential` — disconnect. + +Three rules on those routes: + +1. **The key never comes back out.** A response says a credential is present and + names it; it never returns the secret. A product that cannot read the key + cannot mishandle it, and this is also what keeps the loopback API from becoming + a credential-exfiltration endpoint if something else on the machine reaches it. +2. **The OAuth code stays in the engine.** The product gets the URL to open and an + opaque id, nothing else — the same split `apps/shell/src/main/redrob-connect.ts` + already uses between Electron's main process and its renderer. +3. **No caller-supplied base URL.** Keep this ban. It is not only policy: the + config layer admits a new openai-compatible provider only at a local address + (`isLocalEndpoint`), and honouring an arbitrary URL from a request would + forward the engine's own credential to whatever host the caller named. A local + model is selected by its `provider/model` id. + +### Which providers get a login button + +Ship BYOK for all of them. Add OAuth only where the vendor documents it for third +parties: GitHub Copilot, OpenRouter, Azure OpenAI. Anthropic, Google, OpenAI and +Bedrock get an API-key field and no button. + +The UI should say which is which. A user who expects "sign in with Claude" and +finds a key field deserves the reason in one line — that Anthropic requires an API +key for third-party tools — rather than being left to assume the feature is +missing. + +### Order of work + +1. `GET /v1/providers` and the API-key path. This alone gives Office, Cowork and + Design one shared BYOK setup, and needs no vendor negotiation. +2. OpenRouter OAuth. Smallest real sign-in button and the one that reaches Claude + and GPT models legitimately. +3. GitHub Copilot via the Copilot SDK. A user's existing Copilot subscription, + sanctioned, with a documented device flow for desktop. +4. Azure OpenAI via Entra device code. + +## What must not be built + +- Reusing Claude Code's or the Codex CLI's OAuth client id to spend a consumer + subscription from our own harness. Anthropic and Google both ban accounts for + it; our upstream removed the code under legal pressure. +- Collecting, storing or proxying Claude.ai session tokens. Named explicitly in + Anthropic's compliance page. +- Paying for or intermediating another vendor's usage on a user's behalf. Also + named there, and it is what "just put our key in it" would amount to. + +A compliance check worth keeping: this repository currently contains no Anthropic +OAuth client id and no `claude.ai` endpoint. The only hardcoded vendor OAuth is +xAI's (`packages/redrob/src/plugin/xai.ts`). Keep it that way. From de8fdf3bcf0184152b4c79030a33a55bf17e1b38 Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Tue, 22 Sep 2026 08:34:55 +0000 Subject: [PATCH 4/7] docs: local harness backends for every product, from one place Codex and Claude Code as an option in all ten products from one implementation in this engine, instead of ten adapters in ten repositories. Corrects where the first adapter was put. It was written inside redrob-cowork behind that app's engine-spawn hook; it works and it found real bugs, but it gives Cowork a Codex option and the other nine nothing. It belongs here, because every product already reaches a model through this engine and the ones that still call the Console directly are what /v1/chat/completions was built to bring in. The mechanism turned out to already exist. config/plugin/local-provider.ts admits a provider when the package is exactly the trusted @ai-sdk/openai-compatible and the URL is on this machine - a door opened for Ollama, LM Studio, llama.cpp and vLLM whose only shared trait is speaking OpenAI-compatible chat-completions locally. A small local shim in front of codex exec or claude -p meets the same bar, so this needs no new provider-loading machinery and does not widen the arbitrary-npm-package refusal by a millimetre. This recommends a shim where docs/CODEX-RUNTIME.md argues against one, and the doc says why rather than pretending to be consistent: a Cowork shim must emulate the OpenCode session API, which is large, ours and still changing, so falling subtly behind produces bugs that look like model bugs. A shim here emulates chat-completions, which is small, published, frozen and not ours. Different surface, different answer. What cannot be papered over: these are agent harnesses, not completion endpoints, so support has two tiers. The chat tier - prompt in, text out - covers every product's ordinary use and needs no per-product code. The agent tier, where the runtime owns a real loop against a workspace, cannot go through chat-completions at all, because that contract says tools present means the caller owns them while here the runtime owns the loop. Cowork, code and cad need it; it must not block the chat tier. One safety point that is easy to miss: a harness with a writable sandbox will read and edit the user's filesystem, and a user asking Office to reword a sentence has not consented to an agent walking their disk. The shim pins read-only, approvals never, network off, unless an agent-tier caller asked for more. Sequencing puts /v1/chat/completions first because nothing here ships before it, and records two things to carry rather than rediscover: claude -p's stream-json schema is unverified against the real binary, and Anthropic has announced then paused a change capping third-party subscription usage, so the Claude backend should surface usage state rather than assume it is free. --- docs/LOCAL-HARNESS-BACKENDS.md | 146 +++++++++++++++++++++++++++++++++ 1 file changed, 146 insertions(+) create mode 100644 docs/LOCAL-HARNESS-BACKENDS.md diff --git a/docs/LOCAL-HARNESS-BACKENDS.md b/docs/LOCAL-HARNESS-BACKENDS.md new file mode 100644 index 0000000000..0ffdaec883 --- /dev/null +++ b/docs/LOCAL-HARNESS-BACKENDS.md @@ -0,0 +1,146 @@ +# Local harness backends, for every product at once + +How Codex and Claude Code become an option in *all* Redrob products — office, +cowork, browser, design, extension, canvas, query, recall, cad, reblend — from one +implementation in this engine, rather than ten adapters in ten repositories. + +Companion documents: `docs/PROVIDER-AUTH.md` (what each vendor permits, and why we +never hold a token) and `docs/LOCAL-ENGINE-API.md` (the `/v1/chat/completions` +contract this rides on). redrob-cowork `docs/CODEX-RUNTIME.md` holds the working +prototype that proved the subprocess mechanics. + +## The mistake this corrects + +The first adapter was written inside redrob-cowork, behind that app's engine-spawn +hook. It works, and it found real bugs, but its home is wrong: it gives Cowork a +Codex option and gives the other nine products nothing. Ten products would mean +ten adapters, ten settings screens, ten credential stories, and ten places for the +subprocess environment bug described below to be re-introduced. + +The adapter belongs here instead, because every product already reaches a model +through this engine — and the ones that still call the Console directly are exactly +the ones `/v1/chat/completions` was built to bring in. + +## The mechanism already exists + +`packages/core/src/config/plugin/local-provider.ts` admits a provider when two +conditions hold: the package is exactly `@ai-sdk/openai-compatible` (the trusted, +pinned one the Console provider itself uses), and the URL is on this machine or +network. That door was opened for Ollama, LM Studio, llama.cpp and vLLM, whose only +shared trait is that they **speak OpenAI-compatible chat-completions over a local +address**. + +A harness can meet the same bar. Put a small local server in front of `codex exec` +or `claude -p` that speaks chat-completions, and it is admissible through a path +this engine already trusts — no new provider-loading machinery, no widening of the +arbitrary-npm-package refusal, no second credential store. + +``` +codex exec / claude -p subprocess; holds its OWN credential + ▲ + harness shim local, OpenAI-compatible, 127.0.0.1 + ▲ + redrob-code engine registered via the existing local-provider path + ▲ + POST /v1/chat/completions one receiving route + ▲ + office · cowork · browser · design · extension · canvas · query · recall · cad · reblend +``` + +Models then select a backend by id — `codex/`, `claude-code/` — which +is the same way a local model is already selected, and needs no new request field. + +### Why a shim here, having argued against one in Cowork + +`docs/CODEX-RUNTIME.md` recommends *against* a protocol shim in Cowork and this +document recommends one. That is not an inconsistency, it is the surface being +different, and the difference is the whole argument. + +A Cowork shim would have to emulate the **OpenCode session API** — sessions, the SSE +event shape, permissions — which is large, ours, and still changing. A shim that +falls subtly behind produces bugs that look like model bugs. A shim here emulates +**chat-completions**, which is small, published, frozen, and not ours to change. The +first is a maintenance liability; the second is an adapter against a stable +contract. + +## Two tiers, and most products only need the first + +This is the part that cannot be papered over: **Codex and Claude Code are agent +harnesses, not completion endpoints.** They run their own loop with their own +tools. So there are two levels of support, not one. + +**Chat tier — all ten products.** Prompt in, text out. The harness's loop runs but +its tools are constrained to nothing the caller did not ask for. This covers every +product's ordinary AI use: rewrite this paragraph, summarise this sheet, answer +this question. It maps cleanly onto `/v1/chat/completions` and needs no per-product +code. + +**Agent tier — cowork, code, cad.** The harness runs a real agent loop against a +workspace. chat-completions cannot express this, because that contract's rule is +"tools present → the caller owns them", and here the *runtime* owns the loop. This +needs ACP or a session surface, and it is separate work. Do not let it block the +chat tier. + +A safety point that belongs in the chat tier and is easy to miss: a harness given a +writable sandbox will happily read and edit the user's filesystem. A user asking +Office to reword a sentence has not consented to an agent walking their disk. So +the shim pins `--sandbox read-only` with approvals set to never, and network access +off, unless a caller is on the agent tier and asked for more. The prototype already +defaults this way; the shim must not relax it for convenience. + +The tool-ownership split surveyed earlier decides which products need more than the +chat tier: + +- **No tool protocol** — query, recall, reblend. Chat tier is the whole story. +- **Host-owned tools** — office, browser, canvas. Chat tier works today. Giving the + harness their tools needs MCP servers (`~/.codex/config.toml`, or per-invocation + `--config mcp_servers.…`) or Codex app-server's `dynamicTools`, which lets the + tool stay in the host process. `dynamicTools` is the better fit and is labelled + experimental by OpenAI, so it is not a foundation to build on yet. +- **Engine-owned tools** — cowork, cad, extension, design. Under a harness backend + the engine's own tools are not in the loop at all. That is the agent tier's + problem to solve. + +## What the products have to do + +Almost nothing, which is the point. + +Nothing at all to *work*: a product that names a model and calls the engine gets +the new backends when the engine gets them. + +One thing to be *usable*: somewhere to turn it on, and three states rather than +one — runtime not installed, installed but not signed in, ready. Collapsing those +into "unavailable" strands the user, because the remedy differs and we are not +allowed to offer the vendor's login ourselves. The remedy we may show is "run +`codex login`" or "run `claude`". + +Credentials need no work anywhere. The harness holds its own; the engine holds none +for it. Cowork already demonstrates the pattern for the BYOK case — it stores no +provider key and treats the engine's `auth.json` as the single source of truth. + +## Sequencing + +1. **Land `/v1/chat/completions`.** It is the receiving route for everything above + and it is not merged: the handler exists on `feature/v1-chat-completions`, and + `test:httpapi` fails without an `httpapi-exercise` scenario. Also needs the + `chatCompletions` capability flag, the SSE OpenAPI patch, and a route test. + Nothing here can ship before it. +2. **Build the shim with both backends.** One local chat-completions server; two + normalizers behind it. The prototype's split — event folding separated from the + subprocess — is what makes the second runtime a second normalizer rather than a + second architecture. Reuse it rather than re-deriving it. +3. **Register through the local-provider path**, and surface the backends in + `/v1/providers` with their three states so a product can render settings + without hardcoding a list. +4. **Verify against real binaries.** Neither runtime is installed on the build + host, so the live path — a real subscription actually paying for a turn — is + unproven until someone runs it on a machine with `codex` and `claude` signed in. +5. **Then the agent tier**, for cowork, code and cad, over ACP. + +Two things to carry forward rather than discover later. `claude -p`'s +`stream-json` event schema has not been checked against the real binary — only the +flags are confirmed — so step 2 starts by reading it, not by assuming it mirrors +Codex's JSONL. And Anthropic has announced, then paused, a change that moves +third-party subscription usage onto a capped monthly credit; it currently still +draws from the subscription, but the trajectory is known, so the Claude backend +should surface usage state rather than assume it is free. From 2f896a2181a212d67ad22411a9f9590b0e3ae7d2 Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:02:21 +0000 Subject: [PATCH 5/7] feat(server): close the CI gates on /v1/chat/completions The route had a handler and a conversion layer but nothing that made CI accept it. This adds the coverage scenario, the capability flag, and route tests - and the tests immediately found a real bug in the handler. The bug: a body that does not decode against ChatCompletionRequest came back as 502 api_error. schemaBodyJson fails with HttpServerError | Schema.SchemaError, and neither was recognised, so both fell through to the catch-all that reports an upstream failure. A caller who sent a malformed payload was told our upstream had failed, which sends them debugging the wrong system. Now classified 400 invalid_request_error. Matched on _tag rather than instanceof, because Schema.SchemaError and the HTTP request errors are separate hierarchies and an instanceof chain over both stops matching silently when either is re-exported through a different module instance. The coverage gate was mandatory and I nearly talked myself out of it. Reading routing.ts, coverage derives from routeKeys(OpenApi.fromApi(modules.PublicApi)) -- the instance httpapi -- while this route is registered on the v2 protocol surface, so I concluded it was out of scope. Running it said otherwise: PublicApi merges the v2 groups, the exerciser reported MISS POST /v1/chat/completions, and --fail-on-missing failed the run. The scenario is real and the reading was wrong. That scenario deliberately exercises the 400 path. A real completion needs a provider credential this harness has none of, so asserting 200 would make the gate depend on the machine it runs on. An unknown model id is refused before any provider resolves, which is deterministic and still covers what most likely breaks: that the body decodes, and that the failure arrives in the OpenAI error envelope rather than Effect's default shape. Verified in effect mode against a real server, not just in coverage mode. The capability flag is chatCompletions: { version: 1, callerTools: true, stream: true } on /experimental/capabilities, with a matching struct in the group schema. Declared as a struct rather than a version number so a caller negotiates a capability instead of pinning an engine build - an app needing caller-owned tools checks callerTools, and an engine without the route omits the key entirely. version is the contract's revision, not the engine's. Five route tests drive the real app in-process through httpapi-layer. The envelope assertions are the point: a caller written against OpenAI reads error.message and error.type, so the tests also assert Effect's { name, data } shape does NOT leak. One test pins the v1 contract by sending every promised field including the caller-owned tools path, and checks the failure is about the model rather than a decode, so dropping a field from the schema fails it. Two things deliberately NOT in this change. The SDK codegen is not regenerated. bun script/generate.ts also runs a formatter across the repo (61 files, mostly markdown table alignment) and its openapi.json output contains routes that are not mine - /api/variant/paraphrase among them - which means the generated SDK was already stale before this branch. Regenerating here would sweep another session's unregenerated surface into this PR. It needs its own commit. This repo has no CI check for stale codegen, which is why it drifted. The SSE response is still undocumented in OpenAPI. The group declares only ChatCompletionResponse, so the streaming path has the same gap /event has and the same fix site in public.ts. Doing it properly means defining a ChatCompletionChunk schema in the protocol, which is more than closing a gate, so it is recorded rather than half-done. Coverage 211 pass / 0 fail / 0 missing. Route tests 5/5. Wire tests 23/23. Typecheck clean on both packages. --- .../instance/httpapi/groups/experimental.ts | 16 ++ .../instance/httpapi/handlers/experimental.ts | 9 +- .../server/httpapi-chat-completion.test.ts | 147 ++++++++++++++++++ .../test/server/httpapi-exercise/index.ts | 29 +++- .../server/src/handlers/chat-completion.ts | 37 ++++- 5 files changed, 228 insertions(+), 10 deletions(-) create mode 100644 packages/redrob/test/server/httpapi-chat-completion.test.ts diff --git a/packages/redrob/src/server/routes/instance/httpapi/groups/experimental.ts b/packages/redrob/src/server/routes/instance/httpapi/groups/experimental.ts index 6101ac16cc..368755fada 100644 --- a/packages/redrob/src/server/routes/instance/httpapi/groups/experimental.ts +++ b/packages/redrob/src/server/routes/instance/httpapi/groups/experimental.ts @@ -25,8 +25,24 @@ const ConsoleStateResponse = Schema.Struct({ switchableOrgCount: NonNegativeInt, }).annotate({ identifier: "ConsoleState" }) +/** + * What this engine's `/v1/chat/completions` supports. + * + * Declared as a struct rather than a version number so a caller negotiates a + * capability instead of pinning an engine build: a downstream app that needs + * caller-owned tools checks `callerTools`, and an older engine that lacks the whole + * route simply omits `chatCompletions`. `version` is the contract's revision, not + * the engine's. + */ +const ChatCompletionsCapability = Schema.Struct({ + version: NonNegativeInt, + callerTools: Schema.Boolean, + stream: Schema.Boolean, +}).annotate({ identifier: "ChatCompletionsCapability" }) + const CapabilitiesResponse = Schema.Struct({ backgroundSubagents: Schema.Boolean, + chatCompletions: ChatCompletionsCapability, }).annotate({ identifier: "ExperimentalCapabilities" }) const ConsoleOrgOption = Schema.Struct({ diff --git a/packages/redrob/src/server/routes/instance/httpapi/handlers/experimental.ts b/packages/redrob/src/server/routes/instance/httpapi/handlers/experimental.ts index b218c6040d..94df91239d 100644 --- a/packages/redrob/src/server/routes/instance/httpapi/handlers/experimental.ts +++ b/packages/redrob/src/server/routes/instance/httpapi/handlers/experimental.ts @@ -37,7 +37,14 @@ export const experimentalHandlers = HttpApiBuilder.group(InstanceHttpApi, "exper const flags = yield* RuntimeFlags.Service const capabilities = Effect.fn("ExperimentalHttpApi.capabilities")(function* () { - return { backgroundSubagents: flags.experimentalBackgroundSubagents } + return { + backgroundSubagents: flags.experimentalBackgroundSubagents, + // Declared so a caller can negotiate instead of pinning an engine version. + // A downstream app reads this, sees what this engine's `/v1/chat/completions` + // supports, and degrades explicitly rather than discovering a gap by getting + // a 400 mid-feature. `version` is the contract revision, not the engine's. + chatCompletions: { version: 1, callerTools: true, stream: true }, + } }) const getConsole = Effect.fn("ExperimentalHttpApi.console")(function* () { diff --git a/packages/redrob/test/server/httpapi-chat-completion.test.ts b/packages/redrob/test/server/httpapi-chat-completion.test.ts new file mode 100644 index 0000000000..25dac31b9f --- /dev/null +++ b/packages/redrob/test/server/httpapi-chat-completion.test.ts @@ -0,0 +1,147 @@ +/** + * Route tests for `POST /v1/chat/completions`, driven in-process against the real + * app through `httpapi-layer`. + * + * Scope note. These tests cover what can be asserted without a provider + * credential: that the route is served on the v2 surface at all, that a body + * decodes against `ChatCompletionRequest`, that every failure comes back in the + * OpenAI error envelope rather than Effect's default shape, and that the v1 + * contract's field names are actually present. A completion that reaches a model + * needs a credential this suite has none of, so the streaming happy path is + * covered by `chat-completion-wire.test.ts` at the conversion layer instead of + * being faked here. + * + * The envelope assertions are the point. A caller written against OpenAI's API + * reads `error.message` and `error.type`; if this route ever returns Effect's + * `{ name, data }` shape instead, every such caller silently sees an + * error-less response and reports a truncated or empty answer rather than a + * failure. + */ +import { describe, expect } from "bun:test" +import { LayerNode } from "@redrob-code/core/effect/layer-node" +import { Effect, Layer } from "effect" +import { HttpClientResponse } from "effect/unstable/http" +import { Session } from "@/session/session" +import { Database } from "@redrob-code/core/database/database" +import { TestInstance } from "../fixture/fixture" +import { testEffect } from "../lib/effect" +import { httpApiLayer, requestInDirectory } from "./httpapi-layer" + +const it = testEffect(Layer.mergeAll(LayerNode.compile(LayerNode.group([Session.node, Database.node])), httpApiLayer)) + +const PATH = "/v1/chat/completions" + +function post(directory: string, body: unknown) { + return requestInDirectory(PATH, directory, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify(body), + }) +} + +function json(response: HttpClientResponse.HttpClientResponse) { + return response.json.pipe(Effect.map((value) => value as T)) +} + +type ErrorEnvelope = { + error?: { message?: unknown; type?: unknown; code?: unknown; param?: unknown } +} + +describe("POST /v1/chat/completions", () => { + it.instance("is served on the v2 surface", () => + Effect.gen(function* () { + const tmp = yield* TestInstance + const response = yield* post(tmp.directory, { + model: "no-such-provider/no-such-model", + messages: [{ role: "user", content: "ping" }], + }) + + // Any status other than 404 proves the route exists and is reached. The + // specific failure is asserted below; this case exists so a registration + // regression (the group dropped from protocol/src/api.ts) fails loudly and + // separately from a behaviour change. + expect(response.status).not.toBe(404) + }), + ) + + it.instance("reports an unknown model in the OpenAI error envelope", () => + Effect.gen(function* () { + const tmp = yield* TestInstance + const response = yield* post(tmp.directory, { + model: "no-such-provider/no-such-model", + messages: [{ role: "user", content: "ping" }], + }) + + expect(response.status).toBe(400) + const body = yield* json(response) + expect(typeof body.error).toBe("object") + expect(typeof body.error?.message).toBe("string") + expect(typeof body.error?.type).toBe("string") + // Effect's default failure shape must NOT leak through. + expect(body).not.toHaveProperty("name") + expect(body).not.toHaveProperty("data") + }), + ) + + it.instance("rejects a body that is not a chat-completions request", () => + Effect.gen(function* () { + const tmp = yield* TestInstance + const response = yield* post(tmp.directory, { model: "x" }) + + expect(response.status).toBe(400) + const body = yield* json(response) + expect(typeof body.error?.message).toBe("string") + }), + ) + + it.instance("rejects an empty message list rather than calling a provider", () => + Effect.gen(function* () { + const tmp = yield* TestInstance + const response = yield* post(tmp.directory, { model: "x/y", messages: [] }) + + expect(response.status).toBe(400) + }), + ) + + // v1 CONTRACT. This is the test that should fail when someone removes or renames + // a field downstream callers read. It asserts the request shape is still + // ACCEPTED (decode succeeds, so the failure is about the model rather than the + // body) for every field the v1 contract promises, including the tools path that + // `docs/LOCAL-ENGINE-API.md` describes as caller-owned. + it.instance("still accepts every field the v1 contract promises", () => + Effect.gen(function* () { + const tmp = yield* TestInstance + const response = yield* post(tmp.directory, { + model: "no-such-provider/no-such-model", + messages: [ + { role: "system", content: "be brief" }, + { role: "user", content: "ping" }, + { role: "assistant", content: "pong" }, + ], + stream: true, + temperature: 0.2, + max_tokens: 64, + tools: [ + { + type: "function", + function: { + name: "get_weather", + description: "Look up the weather", + parameters: { type: "object", properties: { city: { type: "string" } } }, + }, + }, + ], + tool_choice: "auto", + }) + + // 400 for the unknown model, NOT a decode failure: if a promised field had + // been dropped from the schema this would still be 400 but with a decode + // message, so the body is checked for the model rather than the shape. + expect(response.status).toBe(400) + const body = yield* json(response) + const message = String(body.error?.message ?? "") + expect(message.toLowerCase()).not.toContain("expected") + expect(message.toLowerCase()).not.toContain("is missing") + }), + ) +}) diff --git a/packages/redrob/test/server/httpapi-exercise/index.ts b/packages/redrob/test/server/httpapi-exercise/index.ts index 551550e90c..e8cc795e47 100644 --- a/packages/redrob/test/server/httpapi-exercise/index.ts +++ b/packages/redrob/test/server/httpapi-exercise/index.ts @@ -97,9 +97,7 @@ const scenarios: Scenario[] = [ Effect.gen(function* () { object(body) check(body.username === "httpapi-global", "global config update should return patched config") - const text = yield* Effect.promise(() => - Bun.file(path.join(exerciseConfigDirectory, "redrob.jsonc")).text(), - ) + const text = yield* Effect.promise(() => Bun.file(path.join(exerciseConfigDirectory, "redrob.jsonc")).text()) check(text.includes('"username": "httpapi-global"'), "global config update should write isolated config file") }), "status", @@ -581,7 +579,32 @@ const scenarios: Scenario[] = [ http.protected.get("/experimental/capabilities", "experimental.capabilities.get").json(200, (body) => { check(typeof body === "object" && body !== null, "capabilities should be an object") check("backgroundSubagents" in body, "capabilities should report background subagents") + check("chatCompletions" in body, "capabilities should report the chat-completions contract") }), + // The OpenAI-compatible route. Exercised on its 400 path deliberately: a real + // completion needs a provider credential this harness has none of, so asserting + // 200 would make the gate depend on the machine it runs on. An unknown model id + // is refused before any provider is resolved, which is what makes this + // deterministic -- and it still covers the parts most likely to break, namely + // that the body decodes against ChatCompletionRequest and that the failure comes + // back in the OpenAI error envelope rather than Effect's default shape. + http.protected + .post("/v1/chat/completions", "chat.completions") + .at((ctx) => ({ + path: "/v1/chat/completions", + headers: ctx.headers(), + body: { + model: "redrob-httpapi-exercise/no-such-model", + messages: [{ role: "user", content: "ping" }], + }, + })) + .json(400, (body) => { + check(typeof body === "object" && body !== null, "chat completions error should be an object") + const error = (body as { error?: { message?: unknown; type?: unknown } }).error + check(typeof error === "object" && error !== null, "error should be nested under `error`, as OpenAI does") + check(typeof error?.message === "string", "error should carry a message") + check(typeof error?.type === "string", "error should carry a type") + }), http.protected .post("/experimental/session/{sessionID}/background", "experimental.session.background") .mutating() diff --git a/packages/server/src/handlers/chat-completion.ts b/packages/server/src/handlers/chat-completion.ts index dee9c6a36b..da056c4da0 100644 --- a/packages/server/src/handlers/chat-completion.ts +++ b/packages/server/src/handlers/chat-completion.ts @@ -89,10 +89,39 @@ function toWireError(error: unknown): WireError { return { status: 502, message: reason.message, type: "api_error", code: null } } } + // A body that does not decode against ChatCompletionRequest is the CALLER's + // error, so it must not fall through to the 502 catch-all below. + // `schemaBodyJson` fails with `HttpServerError | Schema.SchemaError`, and both + // mean the request never reached a provider — reporting them as `api_error` told + // a caller the upstream had failed when in fact their own payload was malformed, + // which sends them debugging the wrong system. + if (isRequestDecodeError(error)) { + return { + status: 400, + message: error instanceof Error ? error.message : String(error), + type: "invalid_request_error", + code: null, + } + } const message = error instanceof Error ? error.message : String(error) return { status: 502, message, type: "api_error", code: null } } +/** + * Whether this failure happened while reading the request, before any provider was + * involved. + * + * Matched structurally rather than with `instanceof`: `Schema.SchemaError` and the + * HTTP request errors are separate hierarchies, and an `instanceof` chain over both + * silently stops matching when either is re-exported through a different module + * instance. The `_tag` values are part of those errors' public shape. + */ +function isRequestDecodeError(error: unknown): boolean { + if (typeof error !== "object" || error === null) return false + const tag = (error as { _tag?: unknown })._tag + return tag === "SchemaError" || tag === "RequestError" || tag === "HttpServerError" +} + function errorResponse(error: unknown): Effect.Effect { const wire = toWireError(error) return HttpServerResponse.json( @@ -120,9 +149,7 @@ const resolveModel = Effect.fn("chat.resolveModel")(function* (model: string) { const catalog = yield* Catalog.Service const info = yield* catalog.model.get(providerID, modelID) if (info === undefined) { - return yield* Effect.fail( - new ConversionError(`model ${model} is not available on this engine`, "model"), - ) + return yield* Effect.fail(new ConversionError(`model ${model} is not available on this engine`, "model")) } const integrations = yield* Integration.Service @@ -132,9 +159,7 @@ const resolveModel = Effect.fn("chat.resolveModel")(function* (model: string) { // 500 from here would. const connection = yield* integrations.connection.active(Integration.ID.make(providerID)) if (connection !== undefined) { - credential = yield* integrations.connection - .resolve(connection) - .pipe(Effect.catch(() => Effect.succeed(undefined))) + credential = yield* integrations.connection.resolve(connection).pipe(Effect.catch(() => Effect.succeed(undefined))) } return yield* fromCatalogModel(info, credential) }) From 4541e15e1386c84ea56ea0910bf598f7160a4b2f Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:19:38 +0000 Subject: [PATCH 6/7] fix(protocol): stop declaring endpoint errors codegen cannot discriminate CI failed six jobs with: Promise error must have a literal discriminator: server.chat.chat.completions httpapi-codegen requires every declared endpoint error to carry a _tag or name STRING LITERAL to branch on (declaredErrorFields). The four error classes shared one OpenAI envelope and had nothing to discriminate by, so generation threw and every job that generates the client died with it. Removed the error declarations rather than adding a discriminator, for two independent reasons: The handler never used them. It is handleRaw - one request answers with JSON and another with text/event-stream - so it writes every response itself, failures included, through errorResponse(). The declared classes were decoration that had never been on the path. And adding _tag to satisfy the generator would make the generated spec and SDK claim a field the wire does not carry, because OpenAI's envelope has no such key. A spec that lies about the shape is worse than one that is silent about it. So the statuses are documented in the endpoint description instead - 400, 401 with code engine_not_authenticated as the only sign-in trigger, 429, 502 - and the envelope survives as an exported ChatCompletionError type so the handler and any caller build against one shape. docs/LOCAL-ENGINE-API.md already specifies the wire contract. Note this was caught by a real CI gate I had earlier described as absent. The generated client IS checked: packages/client check:generated regenerates and fails on any diff. That is a different check from the SDK staleness I noted in the PR body, and the generated client output is committed here for it. Coverage 211/0/0. Route + wire tests 28/28. protocol, server and client typecheck clean. --- .../client/src/generated-effect/client.ts | 42 ++ packages/client/src/generated/client.ts | 32 + packages/client/src/generated/types.ts | 674 ++++++++++++++++++ .../protocol/src/groups/chat-completion.ts | 69 +- 4 files changed, 781 insertions(+), 36 deletions(-) diff --git a/packages/client/src/generated-effect/client.ts b/packages/client/src/generated-effect/client.ts index a0e209872b..77b6b52a9d 100644 --- a/packages/client/src/generated-effect/client.ts +++ b/packages/client/src/generated-effect/client.ts @@ -705,6 +705,47 @@ const adaptGroup18 = (raw: RawClient["server.projectCopy"]) => ({ refresh: Endpoint18_2(raw), }) +type Endpoint19_0Request = Parameters[0] +type Endpoint19_0Input = { + readonly model: Endpoint19_0Request["payload"]["model"] + readonly messages: Endpoint19_0Request["payload"]["messages"] + readonly tools?: Endpoint19_0Request["payload"]["tools"] + readonly tool_choice?: Endpoint19_0Request["payload"]["tool_choice"] + readonly stream?: Endpoint19_0Request["payload"]["stream"] + readonly max_tokens?: Endpoint19_0Request["payload"]["max_tokens"] + readonly max_completion_tokens?: Endpoint19_0Request["payload"]["max_completion_tokens"] + readonly temperature?: Endpoint19_0Request["payload"]["temperature"] + readonly top_p?: Endpoint19_0Request["payload"]["top_p"] + readonly stop?: Endpoint19_0Request["payload"]["stop"] + readonly seed?: Endpoint19_0Request["payload"]["seed"] + readonly frequency_penalty?: Endpoint19_0Request["payload"]["frequency_penalty"] + readonly presence_penalty?: Endpoint19_0Request["payload"]["presence_penalty"] + readonly reasoning_effort?: Endpoint19_0Request["payload"]["reasoning_effort"] + readonly user?: Endpoint19_0Request["payload"]["user"] +} +const Endpoint19_0 = (raw: RawClient["server.chat"]) => (input: Endpoint19_0Input) => + raw["chat.completions"]({ + payload: { + model: input["model"], + messages: input["messages"], + tools: input["tools"], + tool_choice: input["tool_choice"], + stream: input["stream"], + max_tokens: input["max_tokens"], + max_completion_tokens: input["max_completion_tokens"], + temperature: input["temperature"], + top_p: input["top_p"], + stop: input["stop"], + seed: input["seed"], + frequency_penalty: input["frequency_penalty"], + presence_penalty: input["presence_penalty"], + reasoning_effort: input["reasoning_effort"], + user: input["user"], + }, + }).pipe(Effect.mapError(mapClientError)) + +const adaptGroup19 = (raw: RawClient["server.chat"]) => ({ completions: Endpoint19_0(raw) }) + const adaptClient = (raw: RawClient) => ({ health: adaptGroup0(raw["server.health"]), location: adaptGroup1(raw["server.location"]), @@ -725,6 +766,7 @@ const adaptClient = (raw: RawClient) => ({ questions: adaptGroup16(raw["server.question"]), references: adaptGroup17(raw["server.reference"]), projectCopies: adaptGroup18(raw["server.projectCopy"]), + "server.chat": adaptGroup19(raw["server.chat"]), }) export const make = (options?: { readonly baseUrl?: URL | string }) => diff --git a/packages/client/src/generated/client.ts b/packages/client/src/generated/client.ts index 521b25e0b4..211fad5430 100644 --- a/packages/client/src/generated/client.ts +++ b/packages/client/src/generated/client.ts @@ -116,6 +116,8 @@ import type { ProjectCopiesRemoveOutput, ProjectCopiesRefreshInput, ProjectCopiesRefreshOutput, + ServerChatCompletionsInput, + ServerChatCompletionsOutput, } from "./types" import { ClientError } from "./client-error" @@ -1017,6 +1019,36 @@ export function make(options: ClientOptions) { requestOptions, ), }, + "server.chat": { + completions: (input: ServerChatCompletionsInput, requestOptions?: RequestOptions) => + request( + { + method: "POST", + path: `/v1/chat/completions`, + body: { + model: input["model"], + messages: input["messages"], + tools: input["tools"], + tool_choice: input["tool_choice"], + stream: input["stream"], + max_tokens: input["max_tokens"], + max_completion_tokens: input["max_completion_tokens"], + temperature: input["temperature"], + top_p: input["top_p"], + stop: input["stop"], + seed: input["seed"], + frequency_penalty: input["frequency_penalty"], + presence_penalty: input["presence_penalty"], + reasoning_effort: input["reasoning_effort"], + user: input["user"], + }, + successStatus: 200, + declaredStatuses: [401, 400], + empty: false, + }, + requestOptions, + ), + }, } } diff --git a/packages/client/src/generated/types.ts b/packages/client/src/generated/types.ts index 948d394efd..5db37e3579 100644 --- a/packages/client/src/generated/types.ts +++ b/packages/client/src/generated/types.ts @@ -2881,3 +2881,677 @@ export type ProjectCopiesRefreshInput = { } export type ProjectCopiesRefreshOutput = void + +export type ServerChatCompletionsInput = { + readonly model: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["model"] + readonly messages: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["messages"] + readonly tools?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["tools"] + readonly tool_choice?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["tool_choice"] + readonly stream?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["stream"] + readonly max_tokens?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["max_tokens"] + readonly max_completion_tokens?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["max_completion_tokens"] + readonly temperature?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["temperature"] + readonly top_p?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["top_p"] + readonly stop?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["stop"] + readonly seed?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["seed"] + readonly frequency_penalty?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["frequency_penalty"] + readonly presence_penalty?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["presence_penalty"] + readonly reasoning_effort?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["reasoning_effort"] + readonly user?: { + readonly model: string + readonly messages: ReadonlyArray<{ + readonly role: "system" | "developer" | "user" | "assistant" | "tool" + readonly content?: string | null | ReadonlyArray<{ readonly [x: string]: unknown }> | undefined + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly tool_call_id?: string | undefined + readonly name?: string | undefined + }> + readonly tools?: + | ReadonlyArray<{ + readonly type: "function" + readonly function: { + readonly name: string + readonly description?: string | undefined + readonly parameters?: { readonly [x: string]: unknown } | undefined + } + }> + | undefined + readonly tool_choice?: + | "auto" + | "none" + | "required" + | { readonly type: "function"; readonly function: { readonly name: string } } + | undefined + readonly stream?: boolean | undefined + readonly max_tokens?: number | undefined + readonly max_completion_tokens?: number | undefined + readonly temperature?: number | undefined + readonly top_p?: number | undefined + readonly stop?: string | ReadonlyArray | undefined + readonly seed?: number | undefined + readonly frequency_penalty?: number | undefined + readonly presence_penalty?: number | undefined + readonly reasoning_effort?: string | undefined + readonly user?: string | undefined + }["user"] +} + +export type ServerChatCompletionsOutput = { + readonly id: string + readonly object: "chat.completion" + readonly created: number + readonly model: string + readonly choices: ReadonlyArray<{ + readonly index: number + readonly message: { + readonly role: "assistant" + readonly content: string | null + readonly tool_calls?: + | ReadonlyArray<{ + readonly id: string + readonly type: "function" + readonly function: { readonly name: string; readonly arguments: string } + }> + | undefined + readonly reasoning_content?: string | undefined + } + readonly finish_reason: string | null + }> + readonly usage?: + | { readonly prompt_tokens: number; readonly completion_tokens: number; readonly total_tokens: number } + | undefined +} diff --git a/packages/protocol/src/groups/chat-completion.ts b/packages/protocol/src/groups/chat-completion.ts index 7a4a45a83a..dc1ebdde43 100644 --- a/packages/protocol/src/groups/chat-completion.ts +++ b/packages/protocol/src/groups/chat-completion.ts @@ -133,56 +133,47 @@ export const ChatCompletionResponse = Schema.Struct({ export type ChatCompletionResponse = typeof ChatCompletionResponse.Type /** - * The OpenAI error envelope, so a client's existing error handling works - * unchanged. + * The OpenAI error envelope. * - * `code` is the part clients must branch on. `engine_not_authenticated` means - * the ENGINE has no Console credential, and it is the only condition under which - * a client should offer a sign-in action — the message text is localized - * downstream and is not a contract. Branching on text instead is how a network - * timeout reached a user as "please log in". + * NOT declared on the endpoint, and that is deliberate rather than an omission. + * Two independent reasons: * - * One class per status, because the status is what a client's transport sees - * first and collapsing them would report a rejected key as a bad request. The - * body shape is identical across all of them. + * 1. The handler cannot use declared error schemas. It is `handleRaw`, because + * one request answers with JSON and another with `text/event-stream`, so it + * writes every response itself — including failures, through `errorResponse` + * in `chat-completion.ts`. Declared error classes were never on the path. + * 2. `httpapi-codegen` requires each declared endpoint error to carry a `_tag` + * or `name` STRING LITERAL to discriminate on (`declaredErrorFields`), and + * four statuses sharing one envelope have nothing to discriminate by. + * Adding `_tag` to satisfy it would make the generated spec and SDK claim a + * field the wire does not carry, since OpenAI's envelope has no such key — + * a spec that lies is worse than a spec that is silent. + * + * So the statuses are documented in the endpoint description and specified in + * `docs/LOCAL-ENGINE-API.md`, and this type exists to keep the shape in one place + * for the handler to build against. + * + * `code` is the part clients must branch on. `engine_not_authenticated` means the + * ENGINE has no credential, and it is the only condition under which a client + * should offer a sign-in action — the message text is localized downstream and is + * not a contract. Branching on text instead is how a network timeout reached a + * user as "please log in". */ -const errorFields = { +export const ChatCompletionError = Schema.Struct({ error: Schema.Struct({ message: Schema.String, type: Schema.String, code: Schema.NullOr(Schema.String), param: Schema.optional(Schema.NullOr(Schema.String)), }), -} - -export class ChatBadRequestError extends Schema.ErrorClass("ChatBadRequestError")( - errorFields, - { httpApiStatus: 400 }, -) {} - -/** Rejected or absent credentials. Carries code `engine_not_authenticated`. */ -export class ChatUnauthorizedError extends Schema.ErrorClass("ChatUnauthorizedError")( - errorFields, - { httpApiStatus: 401 }, -) {} - -export class ChatRateLimitError extends Schema.ErrorClass("ChatRateLimitError")( - errorFields, - { httpApiStatus: 429 }, -) {} - -/** Upstream failed, or no route could be resolved for the requested model. */ -export class ChatUpstreamError extends Schema.ErrorClass("ChatUpstreamError")( - errorFields, - { httpApiStatus: 502 }, -) {} +}).annotate({ identifier: "ChatCompletionError" }) +export type ChatCompletionError = typeof ChatCompletionError.Type export const ChatCompletionGroup = HttpApiGroup.make("server.chat") .add( HttpApiEndpoint.post("chat.completions", "/v1/chat/completions", { payload: ChatCompletionRequest, success: ChatCompletionResponse, - error: [ChatBadRequestError, ChatUnauthorizedError, ChatRateLimitError, ChatUpstreamError], }).annotateMerge( OpenApi.annotations({ identifier: "v1.chat.completions", @@ -191,7 +182,13 @@ export const ChatCompletionGroup = HttpApiGroup.make("server.chat") "OpenAI-compatible inference against the engine's own credential and provider registry. " + "Declaring `tools` makes the caller responsible for executing them: the turn ends with " + "finish_reason tool_calls and the results come back as role:tool messages. " + - "`stream: true` returns text/event-stream instead of this JSON body.", + "`stream: true` returns text/event-stream instead of this JSON body. " + + "Failures use OpenAI's error envelope -- {error:{message,type,code,param}} -- with " + + "400 invalid_request_error, 401 authentication_error (code engine_not_authenticated, the " + + "only status a client should offer a sign-in action for), 429 rate_limit_error or " + + "insufficient_quota, and 502 api_error. They are written by the handler rather than " + + "declared as endpoint errors, because this is a raw handler and the envelope carries no " + + "discriminator field for codegen to branch on.", }), ), ) From df76b5a20f466a8865471e573b857cabebb919c3 Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Tue, 22 Sep 2026 09:27:57 +0000 Subject: [PATCH 7/7] test: give the fake LLM server its own router, not the engine's CI failed with: Method 'POST' already declared for route '/v1/chat/completions' taking down every test that stands up both the engine's HttpApi and the fake provider (httpapi-sdk prompt streaming, project-skills prompt context). TestLLMServer piped Layer.provide(HttpRouter.layer), which layer memoization resolves to the SAME HttpRouter instance the engine's HttpApi is built on, so the fake and the app registered into one router. That was invisible while no engine route shared a path with the fake. Adding POST /v1/chat/completions to the engine collided with llm-server.ts:702, which serves that exact path. Fixed with Layer.fresh so the fake gets its own router instance. The fake's path is deliberately NOT renamed to dodge the clash. It exists to look like a real OpenAI-compatible endpoint, and /v1/chat/completions is where a real one lives; moving it to a made-up path would stop the fake exercising what the tests are actually about. The collision was a wiring accident, not a naming one, so the wiring is what changed. Worth noting the failure mode: the fake had been sharing the app's router all along, which means any future engine route matching a test double's path would have broken the same way. This makes that class of collision impossible rather than resolving one instance of it. All 130 tests across the 12 files that use TestLLMServer pass (1 pre-existing skip). httpapi-sdk 18/18. --- packages/redrob/test/lib/llm-server.ts | 18 +++++++++++++++++- 1 file changed, 17 insertions(+), 1 deletion(-) diff --git a/packages/redrob/test/lib/llm-server.ts b/packages/redrob/test/lib/llm-server.ts index 245acc7280..6c7f7fbd20 100644 --- a/packages/redrob/test/lib/llm-server.ts +++ b/packages/redrob/test/lib/llm-server.ts @@ -775,5 +775,21 @@ export class TestLLMServer extends Context.Service [...misses]), }) }), - ).pipe(Layer.provide(HttpRouter.layer), Layer.provide(NodeHttpServer.layer(() => Http.createServer(), { port: 0 }))) + ).pipe( + // Layer.fresh, not a bare HttpRouter.layer. Layer memoization otherwise hands + // this fake the SAME HttpRouter instance the engine's own HttpApi is built on, + // so both register into one router. That was harmless only while no engine + // route shared a path with the fake: once the engine gained + // POST /v1/chat/completions -- the real OpenAI path, which this fake must also + // serve to be a credible provider double -- the second registration threw + // "Method 'POST' already declared for route '/v1/chat/completions'" and took + // down every test that stands up both. + // + // The fake's path is deliberately NOT renamed to dodge the clash: it exists to + // look like a real OpenAI-compatible endpoint, and a fake at a made-up path + // would stop exercising the thing under test. Isolating the router is the fix; + // the collision was a wiring accident, not a naming one. + Layer.provide(Layer.fresh(HttpRouter.layer)), + Layer.provide(NodeHttpServer.layer(() => Http.createServer(), { port: 0 })), + ) }