diff --git a/bun.lock b/bun.lock index 1cc1fc02e3..e7c5e9a54b 100644 --- a/bun.lock +++ b/bun.lock @@ -991,6 +991,9 @@ "version": "0.0.0", "dependencies": { "@opencode-ai/client": "workspace:*", + "@opentui/core": "catalog:", + "@opentui/solid": "catalog:", + "solid-js": "catalog:", }, "devDependencies": { "@tsconfig/bun": "catalog:", diff --git a/packages/voice/package.json b/packages/voice/package.json index 917fa23194..25d04af59c 100644 --- a/packages/voice/package.json +++ b/packages/voice/package.json @@ -16,6 +16,9 @@ "typescript": "catalog:" }, "dependencies": { - "@opencode-ai/client": "workspace:*" + "@opencode-ai/client": "workspace:*", + "@opentui/core": "catalog:", + "@opentui/solid": "catalog:", + "solid-js": "catalog:" } } diff --git a/packages/voice/src/spike.ts b/packages/voice/src/spike.ts index 963482c599..abaa93ab4f 100644 --- a/packages/voice/src/spike.ts +++ b/packages/voice/src/spike.ts @@ -29,8 +29,6 @@ const args = parseArgs({ // Needed for full duplex on speakers; harmful with Bluetooth headsets, // where it can bind the wrong capture device. speakers: { type: "boolean", default: false }, - // Start attached to an existing session instead of creating one lazily. - session: { type: "string" }, // Text mode: send one typed message instead of opening the microphone, // print the reply, and exit. Useful for smoke-testing the tool loop. text: { type: "string" }, @@ -62,24 +60,13 @@ const health = await client.health.get().catch((error) => { console.error("or export OPENCODE_SERVER_PASSWORD.") process.exit(1) }) -console.log(`opencode server ${args.server} (version ${health.version})`) -console.log(`project directory: ${args.directory}`) - // --------------------------------------------------------------------------- // OpenCode tool surface exposed to the realtime model // --------------------------------------------------------------------------- -let activeSessionID = args.session +let activeSessionID: string | undefined let lastPromptAt = 0 -if (activeSessionID) { - const session = await client.session.get({ sessionID: activeSessionID }).catch(() => { - console.error(`Session ${activeSessionID} not found on this server.`) - process.exit(1) - }) - console.log(`[voice] controlling session ${session.id} — ${session.title}`) -} - function assistantText(message: SessionMessageAssistant) { return message.content .filter((part) => part.type === "text") @@ -95,44 +82,30 @@ async function requireSession() { if (activeSessionID) return activeSessionID const created = await client.session.create({ location: { directory: args.directory! } }) activeSessionID = created.id - printLine(dim(` [session] created ${activeSessionID}`)) + ui.meta(`[session] created ${activeSessionID}`) + ui.setStatus({ session: activeSessionID }) return created.id } const truncate = (text: string, max = 2000) => (text.length > max ? text.slice(0, max) + "…" : text) // --------------------------------------------------------------------------- -// Terminal output: assistant text streams; everything else must not collide -// with the open streaming line. +// UI: OpenTUI in voice mode, plain console in --text mode. Created before the +// WebSocket so no await sits between socket creation and handler registration. // --------------------------------------------------------------------------- -const tty = process.stdout.isTTY -const dim = (text: string) => (tty ? `\x1b[2m${text}\x1b[0m` : text) -const cyan = (text: string) => (tty ? `\x1b[1;36m${text}\x1b[0m` : text) -const green = (text: string) => (tty ? `\x1b[1;32m${text}\x1b[0m` : text) - -let assistantStreaming = false - -function printLine(line: string) { - if (assistantStreaming) { - process.stdout.write("\n") - assistantStreaming = false - } - console.log(line) -} - -function printAssistantDelta(text: string) { - if (!assistantStreaming) { - process.stdout.write(green("● assistant ") ) - assistantStreaming = true - } - process.stdout.write(text) -} - -function printAssistantDone() { - if (assistantStreaming) process.stdout.write("\n") - assistantStreaming = false -} +const { createConsoleUI, createVoiceTUI } = await import("./ui") +const tuiActive = !args.text && process.stdout.isTTY +const ui = tuiActive + ? await createVoiceTUI({ + onInterrupt: () => interrupt(), + onExit: () => shutdown(), + onCycleVoice: () => cycleVoice(), + }) + : createConsoleUI() +ui.setStatus({ server: args.server }) +ui.meta(`opencode ${args.server} (version ${health.version})`) +ui.meta(`project ${args.directory}`) const toolHandlers: Record) => Promise> = { list_sessions: async (input) => { @@ -153,6 +126,7 @@ const toolHandlers: Record) => Promise { @@ -161,7 +135,7 @@ const toolHandlers: Record) => Promise { @@ -199,6 +173,14 @@ const toolHandlers: Record) => Promise { + const voice = input["voice"] + if (typeof voice !== "string" || !voices.includes(voice)) + return { error: `voice must be one of: ${voices.join(", ")}` } + // Reconnect shortly after replying so the confirmation isn't cut off. + setTimeout(() => setVoice(voice), 1000) + return { voice, note: "Switching requires a brief reconnect; conversation memory resets." } + }, } const toolDefinitions = [ @@ -261,6 +243,22 @@ const toolDefinitions = [ description: "Stop whatever the coding agent is currently doing in the active session.", parameters: { type: "object", properties: {}, required: [] }, }, + { + type: "function", + name: "set_voice", + description: "Change your own speaking voice. Requires a brief reconnect.", + parameters: { + type: "object", + properties: { + voice: { + type: "string", + enum: ["marin", "cedar", "coral", "sage", "ash", "ballad", "alloy", "verse"], + description: "The voice to switch to.", + }, + }, + required: ["voice"], + }, + }, ] const instructions = `You are the voice interface to OpenCode, a coding agent running on the user's machine. @@ -289,6 +287,7 @@ type RealtimeEvent = { type: string delta?: string transcript?: string + item_id?: string item?: RealtimeItem response?: { output?: ReadonlyArray } error?: { type?: string; code?: string; message?: string } @@ -309,13 +308,13 @@ const aecBinary = await (async () => { const binary = Bun.fileURLToPath(new URL("../.build/duplex-audio", import.meta.url)) if ((await Bun.file(binary).exists()) && Bun.file(binary).lastModified > Bun.file(source).lastModified) return binary - console.log("[voice] compiling echo-cancellation helper (first run only)...") + ui.meta("compiling echo-cancellation helper (first run only)...") const { mkdir } = await import("node:fs/promises") await mkdir(Bun.fileURLToPath(new URL("../.build", import.meta.url)), { recursive: true }) const compile = Bun.spawn(["swiftc", "-O", source, "-o", binary], { stdout: "ignore", stderr: "pipe" }) if ((await compile.exited) === 0) return binary - console.error(await new Response(compile.stderr).text()) - console.log("[voice] swiftc failed — falling back to sox audio") + ui.meta(await new Response(compile.stderr).text()) + ui.meta("swiftc failed — falling back to sox audio") return undefined })() @@ -323,13 +322,35 @@ const aecBinary = await (async () => { // voice barge-in is safe on speakers. const fullDuplex = aecBinary !== undefined || args.duplex -const ws = new WebSocket(`wss://api.openai.com/v1/realtime?model=${args.model}`, { - // Bun extension: custom headers on the WebSocket handshake - headers: { Authorization: `Bearer ${apiKey}` }, -} as unknown as string[]) +// Voice can't change once a session has produced audio, so switching voices +// reconnects the realtime socket (conversation context resets; the OpenCode +// session is untouched). +const voices = ["marin", "cedar", "coral", "sage", "ash", "ballad", "alloy", "verse"] +let currentVoice = args.voice ?? "marin" +let ws!: WebSocket +let reconnecting = false const send = (event: Record) => ws.send(JSON.stringify(event)) +function setVoice(voice: string) { + currentVoice = voice + ui.setStatus({ voice }) + ui.meta(`[voice] switching to ${voice}…`) + reconnecting = true + flushPlayback() + ws.close(1000) + connectRealtime() +} + +const cycleVoice = () => setVoice(voices[(voices.indexOf(currentVoice) + 1) % voices.length]!) + +function interrupt() { + if (!assistantSpeaking()) return + send({ type: "response.cancel" }) + flushPlayback() + ui.meta("[interrupted]") +} + let recorder: ReturnType | undefined let player: ReturnType | undefined let audio: ReturnType | undefined // AEC duplex helper (mic + speaker) @@ -342,25 +363,32 @@ const soxFormat = ["-q", "-t", "raw", "-r", "24000", "-e", "signed-integer", "-b let playbackEndsAt = 0 const assistantSpeaking = () => Date.now() < playbackEndsAt +let micStarted = false + async function startMicrophone() { + if (micStarted) return // voice-switch reconnects reuse the running mic + micStarted = true if (aecBinary) { audio = Bun.spawn([aecBinary, ...(args.speakers ? ["--aec"] : [])], { stdin: "pipe", stdout: "pipe", - stderr: args.debug ? "inherit" : "ignore", + stderr: "pipe", }) - console.log("[voice] echo-cancelled duplex audio live — talk any time, even over the assistant (ctrl+c to quit)") + void forwardHelperLogs(audio.stderr as ReadableStream) + ui.setStatus({ audio: args.speakers ? "duplex+aec" : "duplex" }) + ui.meta("mic live — talk any time, even over the assistant") for await (const chunk of audio.stdout as ReadableStream) { - if (ws.readyState !== WebSocket.OPEN) break - send({ type: "input_audio_buffer.append", audio: Buffer.from(chunk).toString("base64") }) + if (ws.readyState === WebSocket.OPEN) + send({ type: "input_audio_buffer.append", audio: Buffer.from(chunk).toString("base64") }) } return } recorder = Bun.spawn(["rec", ...soxFormat, "-"], { stdout: "pipe", stderr: "ignore" }) - console.log("[voice] microphone live — start talking (ctrl+c to quit)") - if (!args.duplex) console.log("[voice] mic mutes while the assistant speaks; press any key to interrupt it") + ui.setStatus({ audio: args.duplex ? "duplex (sox)" : "half-duplex (sox)" }) + ui.meta("mic live — start talking") + if (!args.duplex) ui.meta("mic mutes while the assistant speaks; press any key to interrupt") for await (const chunk of recorder.stdout as ReadableStream) { - if (ws.readyState !== WebSocket.OPEN) break + if (ws.readyState !== WebSocket.OPEN) continue // Half-duplex: drop mic audio while the assistant is audible (plus a // short tail) so speaker echo can't barge-in against itself. if (!args.duplex && Date.now() < playbackEndsAt + 300) continue @@ -368,10 +396,19 @@ async function startMicrophone() { } } +async function forwardHelperLogs(stream: ReadableStream) { + const decoder = new TextDecoder() + for await (const chunk of stream) { + for (const line of decoder.decode(chunk).split("\n")) { + if (line.trim()) ui.meta(line.trim()) + } + } +} + function playAudio(base64: string) { if (!audio) { player ??= Bun.spawn(["play", ...soxFormat, "-"], { stdin: "pipe", stderr: "ignore" }) - if (args.debug) void player.exited.then((code) => console.log(`[debug] play exited (${code})`)) + if (args.debug) void player.exited.then((code) => ui.meta(`[debug] play exited (${code})`)) } const stdin = (audio ?? player)!.stdin as import("bun").FileSink const bytes = Buffer.from(base64, "base64") @@ -399,7 +436,7 @@ async function handleFunctionCall(item: RealtimeItem) { const output = handler ? await handler(JSON.parse(item.arguments ?? "{}")).catch((error) => ({ error: String(error) })) : { error: `unknown tool ${item.name}` } - printLine(dim(` [${item.name}] ${truncate(JSON.stringify(output), 200)}`)) + ui.tool(item.name ?? "unknown", output) send({ type: "conversation.item.create", item: { type: "function_call_output", call_id: item.call_id, output: JSON.stringify(output) }, @@ -408,22 +445,32 @@ async function handleFunctionCall(item: RealtimeItem) { inflightTools -= 1 } -// In half-duplex mode voice barge-in is impossible (the mic is muted while -// the assistant speaks), so any keypress interrupts the assistant instead. -if (!args.text && process.stdin.isTTY) { +// Keypress interrupt for voice mode without the TUI (the TUI handles its own +// keyboard via useKeyboard). +if (!args.text && process.stdin.isTTY && !tuiActive) { process.stdin.setRawMode(true) process.stdin.resume() process.stdin.on("data", (data: Buffer) => { if (data.includes(3)) return shutdown() // ctrl+c - if (!assistantSpeaking()) return - send({ type: "response.cancel" }) - flushPlayback() - printLine(dim(" [interrupted]")) + if (data.toString() === "v") return cycleVoice() + interrupt() }) } -ws.addEventListener("open", () => { - console.log(`[voice] connected to ${args.model}`) +function connectRealtime() { + ws = new WebSocket(`wss://api.openai.com/v1/realtime?model=${args.model}`, { + // Bun extension: custom headers on the WebSocket handshake + headers: { Authorization: `Bearer ${apiKey}` }, + } as unknown as string[]) + ws.addEventListener("open", onOpen) + ws.addEventListener("message", onMessage) + ws.addEventListener("close", onClose) +} + +function onOpen() { + reconnecting = false + ui.meta(`connected to ${args.model} (voice: ${currentVoice})`) + ui.setStatus({ voice: currentVoice }) send({ type: "session.update", session: { @@ -431,7 +478,7 @@ ws.addEventListener("open", () => { instructions, tools: toolDefinitions, tool_choice: "auto", - audio: { output: { voice: args.voice } }, + audio: { output: { voice: currentVoice } }, }, }) // Optional extras sent separately so a shape mismatch can't reject the core session config. @@ -458,18 +505,18 @@ ws.addEventListener("open", () => { }, }, }) -}) +} -ws.addEventListener("message", (event) => { +function onMessage(event: MessageEvent) { const data = JSON.parse(String(event.data)) as RealtimeEvent - if (args.debug && !data.type?.endsWith(".delta")) console.log(`[debug] ${data.type}`) + if (args.debug && !data.type?.endsWith(".delta")) ui.meta(`[debug] ${data.type}`) switch (data.type) { case "session.created": if (!args.text) { void startMicrophone() break } - console.log(`You (typed): ${args.text}`) + ui.userTranscript("typed", args.text) send({ type: "conversation.item.create", item: { type: "message", role: "user", content: [{ type: "input_text", text: args.text }] }, @@ -477,10 +524,10 @@ ws.addEventListener("message", (event) => { createResponse() break case "response.output_text.delta": - printAssistantDelta(data.delta ?? "") + ui.assistantDelta(data.delta ?? "") break case "response.done": { - printAssistantDone() + ui.assistantDone() const calledFunction = data.response?.output?.some((item) => item.type === "function_call") ?? false if (args.text && !calledFunction && inflightTools === 0) shutdown() break @@ -491,33 +538,40 @@ ws.addEventListener("message", (event) => { // playback; keypress is the interrupt. if (fullDuplex) flushPlayback() break + case "input_audio_buffer.committed": + // Reserve the user's slot in the conversation now; the transcript + // arrives later and must not print after the assistant's reply. + ui.userCommitted(data.item_id ?? "") + break case "conversation.item.input_audio_transcription.completed": - printLine(cyan("● you ") + (data.transcript ?? "").trim()) + ui.userTranscript(data.item_id ?? "", (data.transcript ?? "").trim()) break case "response.output_audio.delta": if (data.delta) playAudio(data.delta) break case "response.output_audio_transcript.delta": - printAssistantDelta(data.delta ?? "") + ui.assistantDelta(data.delta ?? "") break case "response.output_audio_transcript.done": - printAssistantDone() + ui.assistantDone() break case "response.output_item.done": if (data.item?.type === "function_call") void handleFunctionCall(data.item) break case "error": - printLine(`[realtime error] ${data.error?.code}: ${data.error?.message}`) + ui.meta(`[realtime error] ${data.error?.code}: ${data.error?.message}`) break } -}) +} -ws.addEventListener("close", (event) => { - console.log(`\n[voice] realtime connection closed (${event.code})`) +function onClose(event: CloseEvent) { + if (reconnecting) return + ui.meta(`realtime connection closed (${event.code})`) shutdown() -}) +} function shutdown() { + ui.close() recorder?.kill() player?.kill() audio?.kill() @@ -525,4 +579,10 @@ function shutdown() { process.exit(0) } +// A surviving process keeps the microphone hot and the OpenAI meter running, +// so every terminal-death signal must tear it down. process.on("SIGINT", shutdown) +process.on("SIGHUP", shutdown) +process.on("SIGTERM", shutdown) + +connectRealtime() diff --git a/packages/voice/src/ui.tsx b/packages/voice/src/ui.tsx new file mode 100644 index 0000000000..a6eafafb23 --- /dev/null +++ b/packages/voice/src/ui.tsx @@ -0,0 +1,246 @@ +/** @jsxImportSource @opentui/solid */ +// Terminal UI for the voice spike. The TUI keeps conversation order stable +// even though realtime events arrive out of order: a user row is inserted the +// moment the audio buffer commits (before the assistant starts replying) and +// its transcript is filled in when Whisper finishes. +import { createCliRenderer } from "@opentui/core" +import { render, useKeyboard } from "@opentui/solid" +import { For } from "solid-js" +import { createStore, produce } from "solid-js/store" + +export type VoiceStatus = { + server?: string + session?: string + audio?: string + voice?: string +} + +export type VoiceUI = { + meta(text: string): void + userCommitted(itemID: string): void + userTranscript(itemID: string, text: string): void + assistantDelta(text: string): void + assistantDone(): void + tool(name: string, output: unknown): void + setStatus(patch: VoiceStatus): void + close(): void +} + +type Message = + | { kind: "user"; itemID: string; text?: string } + | { kind: "assistant"; text: string; streaming: boolean } + | { kind: "tool"; name: string; json: string } + | { kind: "meta"; text: string } + +const truncate = (text: string, max: number) => (text.length > max ? text.slice(0, max) + "…" : text) + +const compactJson = (output: unknown) => + JSON.stringify(output, (_, value) => (typeof value === "string" ? truncate(value, 200) : value)) ?? "null" + +// --------------------------------------------------------------------------- +// Console fallback (--text mode, non-TTY) +// --------------------------------------------------------------------------- + +export function createConsoleUI(): VoiceUI { + const tty = process.stdout.isTTY + const dim = (text: string) => (tty ? `\x1b[2m${text}\x1b[0m` : text) + const cyan = (text: string) => (tty ? `\x1b[1;36m${text}\x1b[0m` : text) + const green = (text: string) => (tty ? `\x1b[1;32m${text}\x1b[0m` : text) + + let streaming = false + const line = (text: string) => { + if (streaming) { + process.stdout.write("\n") + streaming = false + } + console.log(text) + } + + return { + meta: (text) => line(dim(` ${text}`)), + userCommitted: () => {}, + userTranscript: (_, text) => line(cyan("● you ") + text), + assistantDelta: (text) => { + if (!streaming) { + process.stdout.write(green("● assistant ")) + streaming = true + } + process.stdout.write(text) + }, + assistantDone: () => { + if (streaming) process.stdout.write("\n") + streaming = false + }, + tool: (name, output) => line(dim(` [${name}] `) + Bun.inspect(JSON.parse(compactJson(output)), { colors: tty })), + setStatus: () => {}, + close: () => {}, + } +} + +// --------------------------------------------------------------------------- +// OpenTUI +// --------------------------------------------------------------------------- + +const theme = { + text: "#c0caf5", + muted: "#565f89", + you: "#7dcfff", + assistant: "#9ece6a", + key: "#7aa2f7", + string: "#9ece6a", + number: "#e0af68", + literal: "#bb9af7", +} + +// Tiny JSON tokenizer for syntax-highlighted tool results. +function jsonTokens(json: string) { + const tokens: Array<{ text: string; color: string }> = [] + const pattern = /("(?:[^"\\]|\\.)*")(\s*:)?|(-?\d+\.?\d*(?:[eE][+-]?\d+)?)|(true|false|null)|([{}\[\],:]+|\s+)/g + for (const match of json.matchAll(pattern)) { + if (match[1] !== undefined) { + tokens.push({ text: match[1], color: match[2] ? theme.key : theme.string }) + if (match[2]) tokens.push({ text: match[2], color: theme.muted }) + continue + } + if (match[3] !== undefined) { + tokens.push({ text: match[3], color: theme.number }) + continue + } + if (match[4] !== undefined) { + tokens.push({ text: match[4], color: theme.literal }) + continue + } + tokens.push({ text: match[5] ?? "", color: theme.muted }) + } + return tokens +} + +// `kind` never changes after creation, so branching once here is safe; the +// property reads inside JSX stay reactive through the store proxy. +function MessageRow(props: { message: Message }) { + const message = props.message + if (message.kind === "user") + return ( + + ● you {message.text ?? "…"} + + ) + if (message.kind === "assistant") + return ( + + ● assistant {message.text} + {message.streaming ? " …" : ""} + + ) + if (message.kind === "tool") + return ( + + [{message.name}]{" "} + {(token) => {token.text}} + + ) + return ( + + {message.text} + + ) +} + +export async function createVoiceTUI(options: { + onInterrupt(): void + onExit(): void + onCycleVoice(): void +}): Promise { + const [state, setState] = createStore({ + messages: [] as Message[], + status: {} as VoiceStatus, + }) + + const renderer = await createCliRenderer({ + useMouse: true, + exitOnCtrlC: false, + autoFocus: false, + openConsoleOnError: false, + }) + + function App() { + useKeyboard((evt) => { + if (evt.ctrl && evt.name === "c") return options.onExit() + if (evt.name === "v") return options.onCycleVoice() + options.onInterrupt() + }) + const statusLine = () => + [ + state.status.audio ?? "connecting…", + state.status.voice, + state.status.session ?? "no session", + state.status.server, + "v: voice · any key interrupts · ctrl+c quits", + ] + .filter(Boolean) + .join(" ") + return ( + + + + {(message) => } + + + + {statusLine()} + + + ) + } + + void render(() => , renderer) + + const push = (message: Message) => setState("messages", state.messages.length, message) + + return { + meta: (text) => push({ kind: "meta", text }), + userCommitted: (itemID) => push({ kind: "user", itemID }), + userTranscript: (itemID, text) => { + const index = state.messages.findIndex((m) => m.kind === "user" && m.itemID === itemID) + if (index === -1) return push({ kind: "user", itemID, text }) + setState( + "messages", + index, + produce((m) => { + if (m.kind === "user") m.text = text + }), + ) + }, + assistantDelta: (text) => { + const index = state.messages.length - 1 + const last = state.messages[index] + if (last?.kind === "assistant" && last.streaming) { + setState( + "messages", + index, + produce((m) => { + if (m.kind === "assistant") m.text += text + }), + ) + return + } + push({ kind: "assistant", text, streaming: true }) + }, + assistantDone: () => { + const index = state.messages.length - 1 + if (state.messages[index]?.kind !== "assistant") return + setState( + "messages", + index, + produce((m) => { + if (m.kind === "assistant") m.streaming = false + }), + ) + }, + tool: (name, output) => push({ kind: "tool", name, json: compactJson(output) }), + setStatus: (patch) => setState("status", patch), + close: () => { + if (!renderer.isDestroyed) renderer.destroy() + }, + } +} diff --git a/packages/voice/tsconfig.json b/packages/voice/tsconfig.json index 0f2fdf0ce8..accd271c10 100644 --- a/packages/voice/tsconfig.json +++ b/packages/voice/tsconfig.json @@ -2,6 +2,8 @@ "$schema": "https://json.schemastore.org/tsconfig", "extends": "@tsconfig/bun/tsconfig.json", "compilerOptions": { + "jsx": "preserve", + "jsxImportSource": "@opentui/solid", "noUncheckedIndexedAccess": false, "noUnusedLocals": true },