feat: stream Reginald's replies token-by-token
The 26b is slow (~30s/call); the chat now shows the answer forming instead of freezing until it's done. The final prose streams over SSE; tool-calling turns stay structured (no partial tokens), so streaming kicks in for the narration. core (@commitea/core): - chat-client.complete gains an optional onToken — when set, it requests stream:true and parses the OpenAI SSE stream, emitting content deltas and assembling streamed tool-call argument fragments into the final result. - GiteaHttpResponse exposes the optional `body` stream (real fetch has it; stubs don't). agent-loop threads onToken to each completion. app: - model:chat forwards each delta to the renderer (event.sender.send); preload exposes model.onToken(cb) → unsubscribe. useChat accumulates the live stream into a growing bubble (with a cursor), replaced by the authoritative final content when the turn resolves. Unconfigured → scripted reply, unchanged. Verified: 118 core tests green (2 streaming: SSE content deltas + tool-call fragment assembly), desktop typecheck clean, 14 fixture e2e green. Live: a real turn against gemma-4-26b assembles the correct answer via the streaming path (live-reginald green) — the reply now renders token-by-token. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -25,6 +25,48 @@ describe('createChatClient', () => {
|
||||
return { fetch, calls }
|
||||
}
|
||||
|
||||
function sseStub(chunks: string[]) {
|
||||
const calls: { url: string; body: unknown }[] = []
|
||||
const fetch: FetchLike = (url, init) => {
|
||||
calls.push({ url, body: init?.body ? JSON.parse(init.body) : undefined })
|
||||
const enc = new TextEncoder()
|
||||
const body = new ReadableStream<Uint8Array>({
|
||||
start(c) {
|
||||
for (const ch of chunks) c.enqueue(enc.encode(ch))
|
||||
c.close()
|
||||
},
|
||||
})
|
||||
return Promise.resolve({ ok: true, status: 200, body, json: () => Promise.resolve({}), text: () => Promise.resolve('') })
|
||||
}
|
||||
return { fetch, calls }
|
||||
}
|
||||
|
||||
it('streams content deltas via onToken and returns the assembled result', async () => {
|
||||
const { fetch, calls } = sseStub([
|
||||
'data: {"choices":[{"delta":{"content":"Right "}}]}\n\n',
|
||||
'data: {"choices":[{"delta":{"content":"now: #2."}}]}\n\n',
|
||||
'data: [DONE]\n\n',
|
||||
])
|
||||
const client = createChatClient({ baseUrl: 'http://x/v1', model: 'm' }, fetch)
|
||||
const tokens: string[] = []
|
||||
const res = await client.complete([{ role: 'user', content: 'now?' }], undefined, (d) => tokens.push(d))
|
||||
|
||||
expect(tokens).toEqual(['Right ', 'now: #2.'])
|
||||
expect(res.content).toBe('Right now: #2.')
|
||||
expect((calls[0].body as { stream?: boolean }).stream).toBe(true)
|
||||
})
|
||||
|
||||
it('assembles a streamed tool call from argument deltas', async () => {
|
||||
const { fetch } = sseStub([
|
||||
'data: {"choices":[{"delta":{"tool_calls":[{"index":0,"id":"c1","function":{"name":"query_project","arguments":"{\\"view\\""}}]}}]}\n\n',
|
||||
'data: {"choices":[{"delta":{"tool_calls":[{"index":0,"function":{"arguments":":\\"focus\\"}"}}]}}]}\n\n',
|
||||
'data: [DONE]\n\n',
|
||||
])
|
||||
const client = createChatClient({ baseUrl: 'http://x/v1', model: 'm' }, fetch)
|
||||
const res = await client.complete([{ role: 'user', content: 'x' }], [{ name: 'query_project', description: '', parameters: {} }], () => {})
|
||||
expect(res.toolCalls).toEqual([{ id: 'c1', name: 'query_project', arguments: '{"view":"focus"}' }])
|
||||
})
|
||||
|
||||
it('POSTs to /chat/completions and parses content', async () => {
|
||||
const { fetch, calls } = stub({ choices: [{ message: { content: 'the focus is #2' } }] })
|
||||
const client = createChatClient({ baseUrl: 'http://localhost:1234/v1', model: 'gemma' }, fetch)
|
||||
|
||||
Reference in New Issue
Block a user