diff --git a/bun.lock b/bun.lock index 3ea6f31f5d..80cb9ba58a 100644 --- a/bun.lock +++ b/bun.lock @@ -470,7 +470,9 @@ "packages/docs": { "name": "@opencode-ai/docs", "devDependencies": { + "effect": "catalog:", "mint": "4.2.666", + "prettier": "3.6.2", }, }, "packages/effect-drizzle-sqlite": { @@ -1005,6 +1007,24 @@ "@typescript/native-preview": "catalog:", }, }, + "packages/voice": { + "name": "@opencode-ai/voice", + "version": "0.0.0", + "dependencies": { + "@opencode-ai/client": "workspace:*", + "@opentui/core": "catalog:", + "@opentui/solid": "catalog:", + "effect": "catalog:", + "opentui-spinner": "catalog:", + "solid-js": "1.9.12", + }, + "devDependencies": { + "@tsconfig/bun": "catalog:", + "@types/bun": "catalog:", + "@typescript/native-preview": "catalog:", + "typescript": "catalog:", + }, + }, "packages/web": { "name": "@opencode-ai/web", "version": "1.18.4", @@ -2137,6 +2157,8 @@ "@opencode-ai/util": ["@opencode-ai/util@workspace:packages/util"], + "@opencode-ai/voice": ["@opencode-ai/voice@workspace:packages/voice"], + "@opencode-ai/web": ["@opencode-ai/web@workspace:packages/web"], "@opencode-ai/www": ["@opencode-ai/www@workspace:packages/www"], @@ -6605,6 +6627,8 @@ "@opencode-ai/updates/wrangler": ["wrangler@4.110.0", "", { "dependencies": { "@cloudflare/kv-asset-handler": "0.5.0", "@cloudflare/unenv-preset": "2.16.1", "blake3-wasm": "2.1.5", "esbuild": "0.28.1", "miniflare": "4.20260708.1", "path-to-regexp": "6.3.0", "unenv": "2.0.0-rc.24", "workerd": "1.20260708.1" }, "optionalDependencies": { "fsevents": "2.3.3" }, "peerDependencies": { "@cloudflare/workers-types": "^5.20260708.1" }, "optionalPeers": ["@cloudflare/workers-types"], "bin": { "wrangler": "bin/wrangler.js", "wrangler2": "bin/wrangler.js", "cf-wrangler": "bin/cf-wrangler.js" } }, "sha512-xZeXKYi7hxQRF5anL+v77RkufJNpF9f3Eqeyqq2QBsETpLZgh0Agj0jJ6JPtkbgn6ukZdh8OK5egsGPWIditgg=="], + "@opencode-ai/voice/solid-js": ["solid-js@1.9.12", "", { "dependencies": { "csstype": "^3.1.0", "seroval": "~1.5.0", "seroval-plugins": "~1.5.0" } }, "sha512-QzKaSJq2/iDrWR1As6MHZQ8fQkdOBf8GReYb7L5iKwMGceg7HxDcaOHk0at66tNgn9U2U7dXo8ZZpLIAmGMzgw=="], + "@opencode-ai/web/@shikijs/transformers": ["@shikijs/transformers@3.20.0", "", { "dependencies": { "@shikijs/core": "3.20.0", "@shikijs/types": "3.20.0" } }, "sha512-PrHHMRr3Q5W1qB/42kJW6laqFyWdhrPF2hNR9qjOm1xcSiAO3hAHo7HaVyHE6pMyevmy3i51O8kuGGXC78uK3g=="], "@opencode-ai/www/@cloudflare/vite-plugin": ["@cloudflare/vite-plugin@1.44.0", "", { "dependencies": { "@cloudflare/unenv-preset": "2.16.1", "miniflare": "4.20260708.1", "unenv": "2.0.0-rc.24", "wrangler": "4.110.0", "ws": "8.21.0" }, "peerDependencies": { "vite": "^6.1.0 || ^7.0.0 || ^8.0.0" }, "bin": { "cf-vite": "bin/cf-vite" } }, "sha512-8wGGunqRcs34o4GRq0Rurp7GZg30xtLJeRGUU81a49r9zQRjlp3xIlsWr3nFlSCso4eE3cjZfiKC/2y116M4TQ=="], @@ -7885,6 +7909,10 @@ "@opencode-ai/updates/wrangler/workerd": ["workerd@1.20260708.1", "", { "optionalDependencies": { "@cloudflare/workerd-darwin-64": "1.20260708.1", "@cloudflare/workerd-darwin-arm64": "1.20260708.1", "@cloudflare/workerd-linux-64": "1.20260708.1", "@cloudflare/workerd-linux-arm64": "1.20260708.1", "@cloudflare/workerd-windows-64": "1.20260708.1" }, "bin": { "workerd": "bin/workerd" } }, "sha512-WAK+Kt/VVCSldH2qSr8lx46XCJ4Q+bdlHNaFqUtOHthBEIB8C1N8HVW+VOLrxDoTCk0NGNv0zajnBeQK4JOB9w=="], + "@opencode-ai/voice/solid-js/seroval": ["seroval@1.5.5", "", {}, "sha512-bSjOuPcwPKLSJNhr9+bZxA20nQxVle5J5MNsYRVE6cIg7KpRLXGupymePavu0jrxlPiPsr4xGZSB8yUY2sH2sw=="], + + "@opencode-ai/voice/solid-js/seroval-plugins": ["seroval-plugins@1.5.5", "", { "peerDependencies": { "seroval": "^1.0" } }, "sha512-+BDhqYM6CEn3x09v44dpa9p6974FuUB2dxk+Ctn04k0cO1Zt6QODTXfmEZK0eBaTe/fJBvP4NMGuNJ+R8T+QMg=="], + "@opencode-ai/web/@shikijs/transformers/@shikijs/core": ["@shikijs/core@3.20.0", "", { "dependencies": { "@shikijs/types": "3.20.0", "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4", "hast-util-to-html": "^9.0.5" } }, "sha512-f2ED7HYV4JEk827mtMDwe/yQ25pRiXZmtHjWF8uzZKuKiEsJR7Ce1nuQ+HhV9FzDcbIo4ObBCD9GPTzNuy9S1g=="], "@opencode-ai/web/@shikijs/transformers/@shikijs/types": ["@shikijs/types@3.20.0", "", { "dependencies": { "@shikijs/vscode-textmate": "^10.0.2", "@types/hast": "^3.0.4" } }, "sha512-lhYAATn10nkZcBQ0BlzSbJA3wcmL5MXUUF8d2Zzon6saZDlToKaiRX60n2+ZaHJCmXEcZRWNzn+k9vplr8Jhsw=="], diff --git a/packages/client/src/promise/service.ts b/packages/client/src/promise/service.ts index 6d8aa5e327..ee9056e669 100644 --- a/packages/client/src/promise/service.ts +++ b/packages/client/src/promise/service.ts @@ -37,7 +37,6 @@ export async function ensure(options: EnsureOptions = {}): Promise { let announced = false let lastSpawn = 0 let spawnDelay = 5_000 - let ownerHeld = false const announce = (reason: "missing" | "version-mismatch", previousVersion?: string) => { if (announced) return @@ -65,7 +64,6 @@ export async function ensure(options: EnsureOptions = {}): Promise { const registration = await registered(options.file, true) if (registration.service !== undefined) { - ownerHeld = false spawnDelay = 5_000 const service = registration.service const compatible = !service.legacy && (options.version === undefined || service.version === options.version) @@ -82,7 +80,6 @@ export async function ensure(options: EnsureOptions = {}): Promise { if (failure !== undefined) throw failure const finished = [...contenders].filter(contenderFinished) if (finished.some((item) => item.child.exitCode === 0)) { - ownerHeld = true spawnDelay = Math.min(spawnDelay * 2, 30_000) } finished.forEach((item) => contenders.delete(item)) diff --git a/packages/client/test/promise.test.ts b/packages/client/test/promise.test.ts index a9d25ed1c8..32c47ff2be 100644 --- a/packages/client/test/promise.test.ts +++ b/packages/client/test/promise.test.ts @@ -427,6 +427,7 @@ test("session methods use the public HTTP contract", async () => { sessionID: "ses_test", model: { id: "claude", providerID: "anthropic" }, }) + await client.session.archive({ sessionID: "ses_test" }) const admitted = await client.session.prompt({ sessionID: "ses_test", text: "Hello", @@ -462,6 +463,7 @@ test("session methods use the public HTTP contract", async () => { ["POST", "http://localhost:3000/api/session"], ["POST", "http://localhost:3000/api/session/ses_test/agent"], ["POST", "http://localhost:3000/api/session/ses_test/model"], + ["POST", "http://localhost:3000/api/session/ses_test/archive"], ["POST", "http://localhost:3000/api/session/ses_test/prompt"], ["POST", "http://localhost:3000/api/session/ses_test/generate"], ["POST", "http://localhost:3000/api/session/ses_test/synthetic"], diff --git a/packages/core/test/session-create.test.ts b/packages/core/test/session-create.test.ts index a34e9121fc..4ba3632037 100644 --- a/packages/core/test/session-create.test.ts +++ b/packages/core/test/session-create.test.ts @@ -676,4 +676,17 @@ describe("SessionV2.create", () => { expect(events.filter((event) => event.type === "session.archived")).toHaveLength(1) }), ) + + it.effect("rejects archiving a missing Session", () => + Effect.gen(function* () { + const session = yield* SessionV2.Service + + expect( + yield* session.archive(SessionV2.ID.make("ses_missing_archive")).pipe( + Effect.flip, + Effect.map((error) => error._tag), + ), + ).toBe("Session.NotFoundError") + }), + ) }) diff --git a/packages/voice/.gitignore b/packages/voice/.gitignore new file mode 100644 index 0000000000..30bcfa4ed5 --- /dev/null +++ b/packages/voice/.gitignore @@ -0,0 +1 @@ +.build/ diff --git a/packages/voice/bunfig.toml b/packages/voice/bunfig.toml new file mode 100644 index 0000000000..b16283cb5b --- /dev/null +++ b/packages/voice/bunfig.toml @@ -0,0 +1,4 @@ +preload = ["@opentui/solid/preload"] + +[test] +preload = ["@opentui/solid/preload"] diff --git a/packages/voice/package.json b/packages/voice/package.json new file mode 100644 index 0000000000..e280e46f4a --- /dev/null +++ b/packages/voice/package.json @@ -0,0 +1,27 @@ +{ + "$schema": "https://json.schemastore.org/package.json", + "name": "@opencode-ai/voice", + "version": "0.0.0", + "private": true, + "description": "Prototype voice control for OpenCode via the OpenAI Realtime and Live APIs", + "type": "module", + "scripts": { + "spike": "bun run --no-orphans --conditions=browser src/spike.ts", + "test": "bun test", + "typecheck": "tsgo --noEmit" + }, + "devDependencies": { + "@tsconfig/bun": "catalog:", + "@types/bun": "catalog:", + "@typescript/native-preview": "catalog:", + "typescript": "catalog:" + }, + "dependencies": { + "@opencode-ai/client": "workspace:*", + "@opentui/core": "catalog:", + "@opentui/solid": "catalog:", + "effect": "catalog:", + "opentui-spinner": "catalog:", + "solid-js": "1.9.12" + } +} diff --git a/packages/voice/src/animation.ts b/packages/voice/src/animation.ts new file mode 100644 index 0000000000..061da2c9f6 --- /dev/null +++ b/packages/voice/src/animation.ts @@ -0,0 +1,33 @@ +export type TextReveal = { readonly offset: number; readonly at: number } + +const REVEAL_DURATION_MS = 320 +const REVEAL_STAGGER_MS = 32 +const REVEAL_MAX_QUEUE_MS = 160 +export const REVEAL_WORD_LIMIT = 24 + +export function springOpacity(now: number, start: number) { + const progress = Math.max(0, Math.min(1, (now - start) / REVEAL_DURATION_MS)) + if (progress === 1) return 1 + const time = progress * 8 + return 1 - (1 + time) * Math.exp(-time) +} + +export function transcriptionPulse(now: number) { + return 0.5 + Math.sin(now / 220) * 0.5 +} + +export function scheduleTextReveal(previous: string, delta: string, now: number, revealAt: number) { + const text = previous + delta + const offsets = Array.from(delta.matchAll(/\S+/g)) + .map((match) => previous.length + match.index) + .filter((offset) => offset === 0 || /\s/.test(text[offset - 1] ?? "")) + .slice(-REVEAL_WORD_LIMIT) + const start = Math.min(Math.max(revealAt, now), now + REVEAL_MAX_QUEUE_MS) + const reveals = offsets.map((offset, word) => ({ offset, at: start + word * REVEAL_STAGGER_MS })) + const nextRevealAt = reveals.length === 0 ? revealAt : reveals.at(-1)!.at + REVEAL_STAGGER_MS + return { + reveals, + nextRevealAt, + animationEndsAt: reveals.length === 0 ? now : reveals.at(-1)!.at + REVEAL_DURATION_MS, + } +} diff --git a/packages/voice/src/audio-jitter-buffer.ts b/packages/voice/src/audio-jitter-buffer.ts new file mode 100644 index 0000000000..ca60eb53a0 --- /dev/null +++ b/packages/voice/src/audio-jitter-buffer.ts @@ -0,0 +1,41 @@ +import { PCM_BYTES_PER_MS } from "./pcm" + +export type PlaybackChunk = { + readonly bytes: Buffer + readonly gapMs: number +} + +export class AudioJitterBuffer { + private readonly pending: PlaybackChunk[] = [] + private durationMs = 0 + private started = false + + constructor(private readonly targetMs = 500) {} + + push(chunk: PlaybackChunk): ReadonlyArray { + if (this.started) return [chunk] + this.pending.push(chunk) + this.durationMs += chunk.gapMs + chunk.bytes.length / PCM_BYTES_PER_MS + if (this.durationMs < this.targetMs) return [] + this.started = true + return this.drain() + } + + finish(): ReadonlyArray { + if (this.pending.length === 0) return [] + this.started = true + return this.drain() + } + + reset() { + this.pending.length = 0 + this.durationMs = 0 + this.started = false + } + + private drain() { + const chunks = this.pending.splice(0) + this.durationMs = 0 + return chunks + } +} diff --git a/packages/voice/src/completion-store.ts b/packages/voice/src/completion-store.ts new file mode 100644 index 0000000000..cca359136d --- /dev/null +++ b/packages/voice/src/completion-store.ts @@ -0,0 +1,68 @@ +import { mkdir, rename } from "node:fs/promises" +import { homedir } from "node:os" +import { dirname, join } from "node:path" +import { Option, Schema } from "effect" +import { OpenCodeNotification } from "./opencode-notification" + +export type CompletionHandle = { readonly sessionID: string; readonly promptID: string } +export type StoredCompletion = + | { readonly status: "admitting"; readonly handle: CompletionHandle; readonly text: string } + | { readonly status: "pending"; readonly handle: CompletionHandle } + | { + readonly status: "completed" + readonly handle: CompletionHandle + readonly notification: OpenCodeNotification + } + +const CompletionHandle = Schema.Struct({ sessionID: Schema.String, promptID: Schema.String }) +const StoredCompletion = Schema.Union([ + Schema.Struct({ status: Schema.Literal("admitting"), handle: CompletionHandle, text: Schema.String }), + Schema.Struct({ status: Schema.Literal("pending"), handle: CompletionHandle }), + Schema.Struct({ status: Schema.Literal("completed"), handle: CompletionHandle, notification: OpenCodeNotification }), +]) +const StoredCompletions = Schema.Array(StoredCompletion) + +export async function createCompletionStore( + path = join(process.env["XDG_STATE_HOME"] ?? join(homedir(), ".local", "state"), "opencode", "voice-prompts.json"), +) { + await mkdir(dirname(path), { recursive: true }) + const decoded = (await Bun.file(path).exists()) + ? Option.getOrUndefined(Schema.decodeUnknownOption(StoredCompletions)(await Bun.file(path).json())) + : [] + if (!decoded) throw new Error(`Invalid voice completion store: ${path}`) + const entries = new Map(decoded.map((entry) => [entry.handle.promptID, entry])) + let writes = Promise.resolve() + + const save = () => { + const json = JSON.stringify([...entries.values()]) + writes = writes.then(async () => { + const temporary = `${path}.${process.pid}.tmp` + await Bun.write(temporary, json) + await rename(temporary, path) + }) + return writes + } + + return { + entries: () => [...entries.values()], + admitting(handle: CompletionHandle, text: string) { + entries.set(handle.promptID, { status: "admitting", handle, text }) + return save() + }, + pending(handle: CompletionHandle) { + entries.set(handle.promptID, { status: "pending", handle }) + return save() + }, + completed(handle: CompletionHandle, notification: OpenCodeNotification) { + entries.set(handle.promptID, { status: "completed", handle, notification }) + return save() + }, + delivered(promptID: string) { + if (!entries.delete(promptID)) return writes + return save() + }, + close: () => writes, + } +} + +export type CompletionStore = Awaited> diff --git a/packages/voice/src/duplex-audio.swift b/packages/voice/src/duplex-audio.swift new file mode 100644 index 0000000000..06bf051108 --- /dev/null +++ b/packages/voice/src/duplex-audio.swift @@ -0,0 +1,215 @@ +// Full-duplex terminal audio bridge with Apple voice processing (AEC). +// +// stdin <- raw PCM16 mono 24kHz to play through the speakers +// stdout -> raw PCM16 mono 24kHz captured from the microphone +// SIGUSR1: drop any queued speaker audio (barge-in flush) +// +// Two independent engines: attaching a speaker source to a voice-processed +// engine silently kills its input tap, so playback runs on its own plain +// engine. Voice processing (echo cancellation) is attempted on the input +// engine and abandoned if the mic delivers nothing: on some devices +// (Bluetooth headsets mid-negotiation) the VP tap never fires. Without VP +// there is no echo cancellation, which is acceptable exactly when it happens: +// headphones have no echo path. +// +// Compiled on demand by spike.ts: swiftc -O duplex-audio.swift -o duplex-audio + +import AVFoundation + +let stderr = FileHandle.standardError +let sampleRate = 24000.0 +let speakerCapacity = Int(sampleRate) * 120 +func log(_ message: String) { stderr.write(Data("[audio] \(message)\n".utf8)) } + +final class SpeakerQueue { + private let capacity = speakerCapacity + private var samples = [Int16](repeating: 0, count: speakerCapacity) + private var readIndex = 0 + private var count = 0 + private let lock = NSLock() + + func push(_ chunk: Data) { + chunk.withUnsafeBytes { raw in + let input = raw.bindMemory(to: Int16.self) + lock.lock() + for sample in input { + if count == capacity { + samples[readIndex] = sample + readIndex = (readIndex + 1) % capacity + } else { + samples[(readIndex + count) % capacity] = sample + count += 1 + } + } + lock.unlock() + } + } + + func fill(_ output: UnsafeMutablePointer, frames: Int) { + lock.lock() + defer { lock.unlock() } + let available = min(frames, count) + for index in 0.. OSStatus in + let out = UnsafeMutableAudioBufferListPointer(audioBufferList)[0].mData!.assumingMemoryBound(to: Float.self) + queue.fill(out, frames: Int(frameCount)) + return noErr + } + engine.attach(source) + engine.connect(source, to: engine.mainMixerNode, format: playFormat) + do { + try engine.start() + } catch { + log("output engine failed: \(error)") + } +} + +// Microphone: take channel 0 of whatever the hardware provides, resample to +// 24kHz PCM16 for stdout. Channel extraction is manual because hardware +// channel counts vary wildly (1, 2, 3, 22...) and AVAudioConverter cannot +// downmix all of them. +var tapCount = 0 +var inputEngine: AVAudioEngine? + +func startInput(voiceProcessing: Bool) { + inputEngine?.stop() + let engine = AVAudioEngine() + inputEngine = engine + if voiceProcessing { + do { + try engine.inputNode.setVoiceProcessingEnabled(true) + } catch { + log("voice processing unavailable (\(error)) — no echo cancellation") + return startInput(voiceProcessing: false) + } + if #available(macOS 14.0, *) { + engine.inputNode.voiceProcessingOtherAudioDuckingConfiguration = .init( + enableAdvancedDucking: false, + duckingLevel: .min + ) + } + } + + let micFormat = engine.inputNode.outputFormat(forBus: 0) + let monoFormat = AVAudioFormat( + commonFormat: .pcmFormatFloat32, sampleRate: micFormat.sampleRate, channels: 1, interleaved: false)! + guard let converter = AVAudioConverter(from: monoFormat, to: captureFormat) else { + log("cannot convert \(micFormat.sampleRate)Hz to 24kHz") + exit(2) + } + + engine.inputNode.installTap(onBus: 0, bufferSize: 2400, format: micFormat) { buffer, _ in + tapCount += 1 + if tapCount == 1 { log("mic active (echo cancellation \(voiceProcessing ? "on" : "off"))") } + guard let channel = buffer.floatChannelData?[0], buffer.frameLength > 0 else { return } + guard let mono = AVAudioPCMBuffer(pcmFormat: monoFormat, frameCapacity: buffer.frameLength) else { return } + memcpy(mono.floatChannelData![0], channel, Int(buffer.frameLength) * 4) + mono.frameLength = buffer.frameLength + + let capacity = AVAudioFrameCount(Double(buffer.frameLength) * sampleRate / micFormat.sampleRate) + 32 + guard let converted = AVAudioPCMBuffer(pcmFormat: captureFormat, frameCapacity: capacity) else { return } + var consumed = false + converter.convert(to: converted, error: nil) { _, status in + if consumed { + status.pointee = .noDataNow + return nil + } + consumed = true + status.pointee = .haveData + return mono + } + guard converted.frameLength > 0, let out = converted.int16ChannelData?[0] else { return } + FileHandle.standardOutput.write(Data(bytes: out, count: Int(converted.frameLength) * 2)) + } + + do { + try engine.start() + } catch { + if voiceProcessing { + log("input engine failed with voice processing (\(error)) — retrying without") + return startInput(voiceProcessing: false) + } + log("input engine failed: \(error)") + exit(2) + } + + // Watchdog: on some devices the voice-processed tap simply never fires. + // Fall back to a plain tap; without VP the mic reliably delivers. + if voiceProcessing { + DispatchQueue.main.asyncAfter(deadline: .now() + 2.5) { + if tapCount > 0 { return } + log("voice-processed mic delivered nothing — restarting without echo cancellation") + startInput(voiceProcessing: false) + } + } +} + +// Voice processing is opt-in (--aec): it is only needed on speakers, and on +// some machines (observed with Bluetooth headsets active) the VP engine binds +// to the wrong capture device entirely, delivering noise instead of the mic. +let wantAEC = CommandLine.arguments.contains("--aec") + +// Mic first: activating the mic flips Bluetooth headsets from music mode to +// headset mode, reconfiguring the output device. Starting output afterwards +// (and rebuilding on any route change below) keeps playback on the live device. +startInput(voiceProcessing: wantAEC) +startOutput() + +// Device switches (Bluetooth profile flips, headphones plugged/unplugged, +// default device changes) stop engines silently. Rebuild both, debounced. +var rebuildScheduled = false +NotificationCenter.default.addObserver( + forName: .AVAudioEngineConfigurationChange, object: nil, queue: .main +) { _ in + if rebuildScheduled { return } + rebuildScheduled = true + DispatchQueue.main.asyncAfter(deadline: .now() + 0.5) { + rebuildScheduled = false + log("audio route changed — rebuilding engines") + startInput(voiceProcessing: wantAEC) + startOutput() + } +} + +signal(SIGUSR1, SIG_IGN) +let flushSignal = DispatchSource.makeSignalSource(signal: SIGUSR1, queue: .main) +flushSignal.setEventHandler { queue.flush() } +flushSignal.resume() + +DispatchQueue.global().async { + while true { + let chunk = FileHandle.standardInput.availableData + if chunk.isEmpty { exit(0) } // parent closed stdin + queue.push(chunk) + } +} + +RunLoop.main.run() diff --git a/packages/voice/src/opencode-notification.ts b/packages/voice/src/opencode-notification.ts new file mode 100644 index 0000000000..d4d074e456 --- /dev/null +++ b/packages/voice/src/opencode-notification.ts @@ -0,0 +1,63 @@ +import { Schema } from "effect" + +const Completed = Schema.Struct({ + type: Schema.Literal("opencode.prompt.completed"), + session_id: Schema.String, + prompt_id: Schema.String, + status: Schema.Literals(["completed", "failed"]), + text: Schema.String, + error: Schema.optional(Schema.String), +}) + +const Failed = Schema.Struct({ + type: Schema.Literal("opencode.prompt.failed"), + session_id: Schema.String, + prompt_id: Schema.String, + status: Schema.Literal("failed"), + error: Schema.String, +}) + +const PermissionBlocked = Schema.Struct({ + type: Schema.Literal("opencode.prompt.blocked"), + prompt_id: Schema.String, + blocker: Schema.Literal("permission"), + session_id: Schema.String, + request_id: Schema.String, + action: Schema.String, + resources: Schema.Unknown, +}) + +const QuestionBlocked = Schema.Struct({ + type: Schema.Literal("opencode.prompt.blocked"), + prompt_id: Schema.String, + blocker: Schema.Literal("question"), + session_id: Schema.String, + request_id: Schema.String, + questions: Schema.Unknown, +}) + +const FormBlocked = Schema.Struct({ + type: Schema.Literal("opencode.prompt.blocked"), + prompt_id: Schema.String, + blocker: Schema.Literal("form"), + session_id: Schema.String, + form_id: Schema.String, + title: Schema.String, + fields: Schema.Unknown, +}) + +const EventsFailed = Schema.Struct({ + type: Schema.Literal("opencode.events.failed"), + error: Schema.String, +}) + +export const OpenCodeNotification = Schema.Union([ + Completed, + Failed, + PermissionBlocked, + QuestionBlocked, + FormBlocked, + EventsFailed, +]) +export type OpenCodeNotification = typeof OpenCodeNotification.Type +export type OpenCodePromptBlocked = Extract diff --git a/packages/voice/src/opencode.ts b/packages/voice/src/opencode.ts new file mode 100644 index 0000000000..76a897e08a --- /dev/null +++ b/packages/voice/src/opencode.ts @@ -0,0 +1,764 @@ +import { realpathSync } from "node:fs" +import { tmpdir } from "node:os" +import { basename, isAbsolute, relative } from "node:path" +import { OpenCode } from "@opencode-ai/client/promise" +import type { SessionMessageAssistant, SessionMessageUser, V2Event } from "@opencode-ai/client/promise" +import { Form, Question, SessionMessage } from "@opencode-ai/client/effect" +import { Context, Effect, FiberMap, Layer, ManagedRuntime, Option, Schema, Stream } from "effect" +import { createCompletionStore, type CompletionStore } from "./completion-store" +import type { OpenCodeNotification, OpenCodePromptBlocked } from "./opencode-notification" +import type { VoiceTool } from "./protocol" + +type Client = ReturnType +type PromptHandle = { readonly sessionID: string; readonly promptID: string } +type Tool = { + readonly description: string + readonly parameters: unknown + readonly execute: (input: Record) => Effect.Effect +} +type BridgeApi = { + readonly definitions: ReadonlyArray + readonly execute: (name: string, input: Record) => Effect.Effect + readonly delivered: (promptID: string) => Effect.Effect + readonly close: Effect.Effect +} + +class Bridge extends Context.Service()("@opencode-ai/voice/OpenCodeBridge") {} + +export async function createOpenCodeBridge(options: { + client: Client + directory: string + model: { readonly providerID: string; readonly id: string; readonly variant?: string } + notify: (notification: OpenCodeNotification) => void + onSession: (sessionID: string) => void + trace?: (event: string, data?: Record) => void + completionStore?: CompletionStore +}) { + const completionStore = options.completionStore ?? (await createCompletionStore()) + const runtime = ManagedRuntime.make(Layer.effect(Bridge, makeBridge({ ...options, completionStore }))) + const bridge = await runtime.runPromise(Bridge) + return { + definitions: bridge.definitions, + execute: (name: string, input: Record) => runtime.runPromise(bridge.execute(name, input)), + delivered: (promptID: string) => runtime.runPromise(bridge.delivered(promptID)), + close: async () => { + await runtime.runPromise(bridge.close) + await completionStore.close() + await Promise.race([runtime.dispose(), Bun.sleep(1_000)]) + }, + } +} + +const makeBridge = Effect.fnUntraced(function* (options: { + client: Client + directory: string + model: { readonly providerID: string; readonly id: string; readonly variant?: string } + notify: (notification: OpenCodeNotification) => void + onSession: (sessionID: string) => void + trace?: (event: string, data?: Record) => void + completionStore: CompletionStore +}) { + const knownProjects = new Set() + const knownSessions = new Set() + const registrations = new Map() + const promoted = new Map() + const latest = new Map() + const announcedBlockers = new Set() + const completions = yield* FiberMap.make() + const eventAbort = yield* Effect.acquireRelease( + Effect.sync(() => new AbortController()), + (controller) => Effect.sync(() => controller.abort()), + ) + + const notify = (value: OpenCodeNotification) => Effect.sync(() => options.notify(value)) + const trace = (event: string, data: Record) => Effect.sync(() => options.trace?.(event, data)) + const request = (run: (signal: AbortSignal) => PromiseLike) => + Effect.tryPromise({ try: run, catch: (cause) => cause }) + const announceBlocker = (key: string, notification: OpenCodeNotification) => { + if (announcedBlockers.has(key)) return Effect.void + announcedBlockers.add(key) + return notify(notification) + } + const clearRegistration = (handle: PromptHandle) => + Effect.sync(() => { + registrations.delete(handle.promptID) + if (promoted.get(handle.sessionID) === handle.promptID) promoted.delete(handle.sessionID) + if (latest.get(handle.sessionID) === handle.promptID) latest.delete(handle.sessionID) + }) + + const listProjects = Effect.fnUntraced(function* () { + const seen = new Set() + const temporary = realpathSync(tmpdir()) + return (yield* request((signal) => options.client.project.list({ signal }))) + .sort((a, b) => b.time.updated - a.time.updated) + .filter((project) => { + const fromTemporary = relative(temporary, project.worktree) + if ( + project.id === "global" || + fromTemporary === "" || + (!fromTemporary.startsWith("..") && !isAbsolute(fromTemporary)) || + seen.has(project.worktree) + ) + return false + seen.add(project.worktree) + return true + }) + }) + + const projectDirectory = Effect.fnUntraced(function* (projectID: string) { + if (!knownProjects.has(projectID)) return undefined + return (yield* listProjects()).find((project) => project.id === projectID)?.worktree + }) + + const complete = Effect.fnUntraced(function* (handle: PromptHandle) { + yield* trace("opencode.wait.started", handle) + yield* request((signal) => options.client.session.wait({ sessionID: handle.sessionID }, { signal })) + const reply = yield* request((signal) => finalReply(options.client, handle, signal)) + yield* trace("opencode.wait.completed", { ...handle, replyID: reply?.id, error: reply?.error?.message }) + const notification: OpenCodeNotification = { + type: "opencode.prompt.completed", + session_id: handle.sessionID, + prompt_id: handle.promptID, + status: reply?.error ? "failed" : "completed", + text: reply + ? assistantText(reply) || "OpenCode finished without a text reply." + : "OpenCode finished without a reply.", + error: reply?.error?.message, + } + yield* request(() => options.completionStore.completed(handle, notification)) + yield* notify(notification) + }) + + const register = Effect.fnUntraced(function* (handle: PromptHandle) { + yield* complete(handle).pipe( + Effect.catch((error) => { + const notification: OpenCodeNotification = { + type: "opencode.prompt.failed", + session_id: handle.sessionID, + prompt_id: handle.promptID, + status: "failed", + error: String(error), + } + return request(() => options.completionStore.completed(handle, notification)).pipe( + Effect.andThen(notify(notification)), + ) + }), + Effect.ensuring(clearRegistration(handle)), + FiberMap.run(completions, handle.promptID, { onlyIfMissing: true, startImmediately: true }), + ) + }) + + const restoreBlockers = Effect.fnUntraced(function* (handle: PromptHandle) { + const [permissions, questions, forms] = yield* Effect.all( + [ + request((signal) => options.client.permission.list({ sessionID: handle.sessionID }, { signal })), + request((signal) => options.client.question.list({ sessionID: handle.sessionID }, { signal })), + request((signal) => options.client.form.list({ sessionID: handle.sessionID }, { signal })), + ], + { concurrency: "unbounded" }, + ) + yield* Effect.forEach( + [ + ...permissions.map((item) => ({ + key: `permission:${item.id}`, + notification: { + type: "opencode.prompt.blocked", + prompt_id: handle.promptID, + blocker: "permission", + session_id: handle.sessionID, + request_id: item.id, + action: item.action, + resources: item.resources, + } satisfies OpenCodeNotification, + })), + ...questions.map((item) => ({ + key: `question:${item.id}`, + notification: { + type: "opencode.prompt.blocked", + prompt_id: handle.promptID, + blocker: "question", + session_id: handle.sessionID, + request_id: item.id, + questions: item.questions, + } satisfies OpenCodeNotification, + })), + ...forms.map((item) => ({ + key: `form:${item.id}`, + notification: { + type: "opencode.prompt.blocked", + prompt_id: handle.promptID, + blocker: "form", + session_id: handle.sessionID, + form_id: item.id, + title: item.title, + fields: item.fields, + } satisfies OpenCodeNotification, + })), + ], + (item) => announceBlocker(item.key, item.notification), + { discard: true }, + ) + }) + + const admit = Effect.fnUntraced(function* (sessionID: string, text: string) { + const handle = { sessionID, promptID: SessionMessage.ID.create() } + registrations.set(handle.promptID, handle) + latest.set(sessionID, handle.promptID) + yield* trace("opencode.prompt.admitting", handle) + yield* request(() => options.completionStore.admitting(handle, text)) + const admitted = yield* request((signal) => + options.client.session.prompt({ sessionID, id: handle.promptID, text }, { signal }), + ).pipe( + Effect.tapError(() => + request(() => options.completionStore.delivered(handle.promptID)).pipe( + Effect.andThen(clearRegistration(handle)), + ), + ), + ) + knownSessions.add(sessionID) + yield* request(() => options.completionStore.pending(handle)) + yield* trace("opencode.prompt.admitted", { ...handle, admittedSeq: admitted.admittedSeq }) + options.onSession(sessionID) + yield* register(handle) + return { + status: "started", + session_id: sessionID, + prompt_id: admitted.id, + notification: "registered", + message: "OpenCode accepted the prompt and will send a completion notification. Continue the conversation now.", + } + }) + + const start = Effect.fnUntraced(function* (text: string, projectID?: string) { + const directory = projectID ? yield* projectDirectory(projectID) : options.directory + if (!directory) return toolError("Use a project ID returned by find_projects.") + const session = yield* request((signal) => + options.client.session.create({ location: { directory }, model: options.model }, { signal }), + ) + return yield* admit(session.id, text) + }) + + const tools: Record = { + find_projects: { + description: "Find known OpenCode projects by display name. Returns opaque IDs, never filesystem paths.", + parameters: { + type: "object", + additionalProperties: false, + properties: { + query: { type: ["string", "null"], description: "Name fragment, or null for recent projects." }, + limit: { type: "integer", minimum: 1, maximum: 20 }, + }, + required: ["query", "limit"], + }, + execute: Effect.fnUntraced(function* (input) { + const query = typeof input["query"] === "string" ? input["query"].toLowerCase() : undefined + const limit = typeof input["limit"] === "number" ? input["limit"] : 10 + const projects = (yield* listProjects()) + .filter((project) => !query || projectLabel(project).toLowerCase().includes(query)) + .slice(0, limit) + projects.forEach((project) => knownProjects.add(project.id)) + return { + status: "ok", + projects: projects.map((project) => ({ + id: project.id, + name: projectLabel(project), + directories: 1 + project.sandboxes.length, + updated: new Date(project.time.updated).toISOString(), + })), + } + }), + }, + find_sessions: { + description: + "Find root OpenCode sessions by title and recency. Search the launch project by default; use all_projects only when requested or the local search misses.", + parameters: { + type: "object", + additionalProperties: false, + properties: { + query: { type: ["string", "null"], description: "Title words, or null for recent sessions." }, + scope: { type: "string", enum: ["current_project", "all_projects"] }, + recency: { type: "string", enum: ["day", "week", "month", "any"] }, + limit: { type: "integer", minimum: 1, maximum: 20 }, + }, + required: ["query", "scope", "recency", "limit"], + }, + execute: Effect.fnUntraced(function* (input) { + const query = typeof input["query"] === "string" ? input["query"] : undefined + const scope = input["scope"] === "all_projects" ? "all_projects" : "current_project" + const recency = typeof input["recency"] === "string" ? input["recency"] : "any" + const limit = typeof input["limit"] === "number" ? input["limit"] : 10 + const durations: Record = { day: 86_400_000, week: 604_800_000, month: 2_592_000_000 } + const threshold = recency === "any" ? 0 : Date.now() - (durations[recency] ?? 0) + const [result, projectList] = yield* Effect.all( + [ + request((signal) => + options.client.session.list( + { + ...(scope === "current_project" ? { directory: options.directory } : {}), + search: query, + parentID: null, + limit: recency === "any" ? limit : Math.min(limit * 5, 100), + order: "desc", + }, + { signal }, + ), + ), + listProjects(), + ], + { concurrency: "unbounded" }, + ) + const projects = new Map(projectList.map((project) => [project.id, projectLabel(project)])) + const sessions = result.data + .filter((session) => projects.has(session.projectID) && session.time.updated >= threshold) + .slice(0, limit) + sessions.forEach((session) => knownSessions.add(session.id)) + return { + status: "ok", + scope, + sessions: sessions.map((session) => ({ + id: session.id, + title: session.title, + project: projects.get(session.projectID) ?? "project", + updated: new Date(session.time.updated).toISOString(), + })), + } + }), + }, + read_session: { + description: + "Read bounded recent user and assistant text from a session returned by find_sessions or start_session. This never prompts or wakes the coding agent.", + parameters: { + type: "object", + additionalProperties: false, + properties: { + session_id: { type: "string", description: "Session ID returned by a previous voice tool." }, + limit: { type: "integer", minimum: 1, maximum: 20 }, + }, + required: ["session_id", "limit"], + }, + execute: Effect.fnUntraced(function* (input) { + const sessionID = knownSession(input, knownSessions) + if (!sessionID) return toolError("Use a session ID returned by find_sessions or start_session.") + const limit = typeof input["limit"] === "number" ? input["limit"] : 10 + const [session, messages] = yield* Effect.all( + [ + request((signal) => options.client.session.get({ sessionID }, { signal })), + request((signal) => options.client.message.list({ sessionID, order: "desc", limit }, { signal })), + ], + { concurrency: "unbounded" }, + ) + const latestAssistant = messages.data.find( + (message): message is SessionMessageAssistant => message.type === "assistant", + ) + return { + status: "ok", + title: session.title, + running: latestAssistant ? !latestAssistant.time.completed : false, + messages: messages.data + .toReversed() + .flatMap((message) => + message.type === "user" || message.type === "assistant" ? [sessionMessage(message)] : [], + ), + } + }), + }, + rename_session: { + description: + "Rename a session returned by find_sessions or start_session after the user requests a new title. This changes only the display title and does not prompt, wake, or interrupt the coding agent.", + parameters: { + type: "object", + additionalProperties: false, + properties: { + session_id: { type: "string", description: "Session ID returned by a previous voice tool." }, + title: { type: "string", description: "The complete new session title." }, + }, + required: ["session_id", "title"], + }, + execute: Effect.fnUntraced(function* (input) { + const sessionID = knownSession(input, knownSessions) + const title = requireString(input, "title")?.trim() + if (!sessionID) return toolError("Use a session ID returned by find_sessions or start_session.") + if (!title) return toolError("A non-empty session title is required.") + const session = yield* request((signal) => options.client.session.get({ sessionID }, { signal })) + if (session.title === title) + return { status: "unchanged", session_id: sessionID, previous_title: session.title, title } + yield* request((signal) => options.client.session.rename({ sessionID, title }, { signal })) + return { status: "renamed", session_id: sessionID, previous_title: session.title, title } + }), + }, + archive_session: { + description: + "Archive a session returned by find_sessions or start_session. Call only after the user explicitly confirms the archive. This changes session visibility and does not prompt, wake, interrupt, or stop the coding agent.", + parameters: sessionParameters(), + execute: Effect.fnUntraced(function* (input) { + const sessionID = knownSession(input, knownSessions) + if (!sessionID) return toolError("Use a session ID returned by find_sessions or start_session.") + const session = yield* request((signal) => options.client.session.get({ sessionID }, { signal })) + if (session.time.archived) return { status: "already_archived", title: session.title } + yield* request((signal) => options.client.session.archive({ sessionID }, { signal })) + return { status: "archived", title: session.title } + }), + }, + start_session: { + description: + "Create an OpenCode session, admit its first prompt, and register a one-shot completion notification. Omit project_id to use the launch project.", + parameters: { + type: "object", + additionalProperties: false, + properties: { + text: { type: "string", description: "Clear instruction for the coding agent." }, + project_id: { type: ["string", "null"], description: "Opaque project ID, or null for the launch project." }, + }, + required: ["text", "project_id"], + }, + execute: (input) => { + const text = requireString(input, "text") + if (!text) return Effect.succeed(toolError("Task text is required.")) + return start(text, typeof input["project_id"] === "string" ? input["project_id"] : undefined) + }, + }, + prompt_session: { + description: + "Admit a prompt into a discovered or previously started OpenCode session and register a one-shot completion notification.", + parameters: { + type: "object", + additionalProperties: false, + properties: { + session_id: { type: "string", description: "Session ID returned by a previous voice tool." }, + text: { type: "string", description: "Clear instruction for the coding agent." }, + }, + required: ["session_id", "text"], + }, + execute: (input) => { + const sessionID = knownSession(input, knownSessions) + const text = requireString(input, "text") + if (!sessionID) return Effect.succeed(toolError("Use a session ID returned by find_sessions or start_session.")) + if (!text) return Effect.succeed(toolError("Task text is required.")) + return admit(sessionID, text) + }, + }, + interrupt_session: { + description: "Interrupt one OpenCode session. Call only after the user explicitly confirms the interruption.", + parameters: sessionParameters(), + execute: Effect.fnUntraced(function* (input) { + const sessionID = knownSession(input, knownSessions) + if (!sessionID) return toolError("Use a session ID returned by a previous voice tool.") + yield* request((signal) => options.client.session.interrupt({ sessionID }, { signal })) + return { status: "interrupted", session_id: sessionID } + }), + }, + list_pending_permissions: { + description: "List permission requests blocking one OpenCode session.", + parameters: sessionParameters(), + execute: Effect.fnUntraced(function* (input) { + const sessionID = knownSession(input, knownSessions) + if (!sessionID) return toolError("Use a session ID returned by a previous voice tool.") + const requests = yield* request((signal) => options.client.permission.list({ sessionID }, { signal })) + return { + status: "ok", + requests: requests.map((item) => ({ id: item.id, action: item.action, resources: item.resources })), + } + }), + }, + reply_permission: { + description: + "Allow once or reject a pending permission after stating the action and resources and receiving the user's explicit decision.", + parameters: { + type: "object", + additionalProperties: false, + properties: { + session_id: { type: "string" }, + request_id: { type: "string" }, + decision: { type: "string", enum: ["allow_once", "reject"] }, + }, + required: ["session_id", "request_id", "decision"], + }, + execute: Effect.fnUntraced(function* (input) { + const sessionID = knownSession(input, knownSessions) + const requestID = requireString(input, "request_id") + const decision = input["decision"] + if (!sessionID || !requestID || (decision !== "allow_once" && decision !== "reject")) + return toolError("A known session, request ID, and valid decision are required.") + const requests = yield* request((signal) => options.client.permission.list({ sessionID }, { signal })) + if (!requests.some((item) => item.id === requestID)) + return toolError("That permission request is not pending.", true) + yield* request((signal) => + options.client.permission.reply( + { sessionID, requestID, reply: decision === "allow_once" ? "once" : "reject" }, + { signal }, + ), + ) + return { status: decision === "allow_once" ? "allowed_once" : "rejected", request_id: requestID } + }), + }, + reply_question: { + description: "Reply to questions blocking an OpenCode session after collecting the user's answers.", + parameters: { + type: "object", + additionalProperties: false, + properties: { + session_id: { type: "string" }, + request_id: { type: "string" }, + answers: { + type: "array", + description: "One string array per question, preserving question order.", + items: { type: "array", items: { type: "string" } }, + }, + }, + required: ["session_id", "request_id", "answers"], + }, + execute: Effect.fnUntraced(function* (input) { + const sessionID = knownSession(input, knownSessions) + const requestID = requireString(input, "request_id") + const answers = Option.getOrUndefined(decodeQuestionAnswers(input["answers"])) + if (!sessionID || !requestID || !answers) + return toolError("A known session, request ID, and valid answers are required.") + yield* request((signal) => options.client.question.reply({ sessionID, requestID, answers }, { signal })) + return { status: "answered", request_id: requestID } + }), + }, + reject_question: { + description: "Reject a pending OpenCode question after the user explicitly declines to answer.", + parameters: requestParameters(), + execute: Effect.fnUntraced(function* (input) { + const sessionID = knownSession(input, knownSessions) + const requestID = requireString(input, "request_id") + if (!sessionID || !requestID) return toolError("A known session and request ID are required.") + yield* request((signal) => options.client.question.reject({ sessionID, requestID }, { signal })) + return { status: "rejected", request_id: requestID } + }), + }, + reply_form: { + description: "Submit values for a form blocking an OpenCode session after collecting them from the user.", + parameters: { + type: "object", + additionalProperties: false, + properties: { + session_id: { type: "string" }, + form_id: { type: "string" }, + answer: { type: "object", additionalProperties: true }, + }, + required: ["session_id", "form_id", "answer"], + }, + execute: Effect.fnUntraced(function* (input) { + const sessionID = knownSession(input, knownSessions) + const formID = requireString(input, "form_id") + const answer = Option.getOrUndefined(decodeFormAnswer(input["answer"])) + if (!sessionID || !formID || !answer) + return toolError("A known session, form ID, and valid answer are required.") + yield* request((signal) => options.client.form.reply({ sessionID, formID, answer }, { signal })) + return { status: "answered", form_id: formID } + }), + }, + cancel_form: { + description: "Cancel a pending OpenCode form after the user explicitly declines to complete it.", + parameters: { + type: "object", + additionalProperties: false, + properties: { session_id: { type: "string" }, form_id: { type: "string" } }, + required: ["session_id", "form_id"], + }, + execute: Effect.fnUntraced(function* (input) { + const sessionID = knownSession(input, knownSessions) + const formID = requireString(input, "form_id") + if (!sessionID || !formID) return toolError("A known session and form ID are required.") + yield* request((signal) => options.client.form.cancel({ sessionID, formID }, { signal })) + return { status: "cancelled", form_id: formID } + }), + }, + } + + const definitions = Object.entries(tools).map( + ([name, tool]) => + ({ type: "function", name, description: tool.description, parameters: tool.parameters }) satisfies VoiceTool, + ) + + yield* Effect.suspend(() => + Stream.fromAsyncIterable(options.client.event.subscribe({ signal: eventAbort.signal }), (cause) => cause).pipe( + Stream.runForEach((event) => { + if (event.type === "session.input.promoted") { + if (!registrations.has(event.data.inputID)) return Effect.void + promoted.set(event.data.sessionID, event.data.inputID) + return trace("opencode.prompt.promoted", { + sessionID: event.data.sessionID, + promptID: event.data.inputID, + }) + } + const sessionID = blockerSession(event) + const promptID = sessionID ? (promoted.get(sessionID) ?? latest.get(sessionID)) : undefined + const blocker = sessionID && promptID ? blockerNotification(event, sessionID, promptID) : undefined + if (!blocker || !promptID) return Effect.void + options.trace?.("opencode.prompt.blocked", { sessionID, promptID, blocker: blocker["blocker"] }) + if (registrations.has(promptID)) return announceBlocker(blockerKey(blocker), blocker) + return Effect.void + }), + Effect.catch((error) => notify({ type: "opencode.events.failed", error: String(error) })), + Effect.andThen(Effect.sleep("1 second")), + ), + ).pipe(Effect.forever, Effect.forkScoped({ startImmediately: true })) + + yield* Effect.forEach( + options.completionStore.entries(), + Effect.fnUntraced(function* (entry) { + knownSessions.add(entry.handle.sessionID) + if (entry.status === "admitting") { + registrations.set(entry.handle.promptID, entry.handle) + latest.set(entry.handle.sessionID, entry.handle.promptID) + yield* trace("opencode.prompt.reconciling", entry.handle) + yield* request((signal) => + options.client.session.prompt( + { sessionID: entry.handle.sessionID, id: entry.handle.promptID, text: entry.text }, + { signal }, + ), + ) + yield* request(() => options.completionStore.pending(entry.handle)) + yield* restoreBlockers(entry.handle) + yield* register(entry.handle) + return + } + if (entry.status === "pending") { + registrations.set(entry.handle.promptID, entry.handle) + latest.set(entry.handle.sessionID, entry.handle.promptID) + yield* trace("opencode.wait.restored", entry.handle) + yield* restoreBlockers(entry.handle) + yield* register(entry.handle) + return + } + yield* trace("opencode.notification.restored", entry.handle) + yield* notify(entry.notification) + }), + { discard: true }, + ) + + return Bridge.of({ + definitions, + execute: (name, input) => tools[name]?.execute(input) ?? Effect.succeed(toolError(`Unknown tool ${name}.`)), + delivered: (promptID) => request(() => options.completionStore.delivered(promptID)), + close: Effect.sync(() => eventAbort.abort()), + }) +}) + +async function finalReply(client: Client, handle: PromptHandle, signal: AbortSignal) { + let reply: SessionMessageAssistant | undefined + let cursor: string | undefined + while (true) { + const page = await client.message.list( + cursor + ? { sessionID: handle.sessionID, limit: 200, cursor } + : { sessionID: handle.sessionID, limit: 200, order: "desc" }, + { signal }, + ) + for (const message of page.data) { + if (message.id === handle.promptID) return reply + if (message.type === "user") reply = undefined + if (message.type === "assistant" && !reply) reply = message + } + cursor = page.cursor.next ?? undefined + if (!cursor) return undefined + } +} + +function assistantText(message: SessionMessageAssistant) { + return message.content + .filter((part) => part.type === "text") + .map((part) => part.text) + .join("\n") + .slice(0, 4_000) +} + +function sessionMessage(message: SessionMessageUser | SessionMessageAssistant) { + if (message.type === "user") return { role: "user", text: message.text.slice(0, 2_000) } + return { + role: "assistant", + status: message.time.completed ? "completed" : "running", + text: assistantText(message), + tools: message.content + .filter((part) => part.type === "tool") + .map((part) => ({ name: part.name, status: part.state.status })), + } +} + +function blockerSession(event: V2Event) { + if (event.type === "permission.v2.asked" || event.type === "question.v2.asked") return event.data.sessionID + if (event.type === "form.created") return event.data.form.sessionID + return undefined +} + +function blockerNotification(event: V2Event, sessionID: string, promptID: string): OpenCodePromptBlocked | undefined { + if (event.type === "permission.v2.asked") + return { + type: "opencode.prompt.blocked", + prompt_id: promptID, + blocker: "permission", + session_id: sessionID, + request_id: event.data.id, + action: event.data.action, + resources: event.data.resources, + } + if (event.type === "question.v2.asked") + return { + type: "opencode.prompt.blocked", + prompt_id: promptID, + blocker: "question", + session_id: sessionID, + request_id: event.data.id, + questions: event.data.questions, + } + if (event.type === "form.created") + return { + type: "opencode.prompt.blocked", + prompt_id: promptID, + blocker: "form", + session_id: sessionID, + form_id: event.data.form.id, + title: event.data.form.title, + fields: event.data.form.fields, + } + return undefined +} + +function blockerKey(notification: OpenCodePromptBlocked) { + if (notification.blocker === "form") return `form:${notification.form_id}` + return `${notification.blocker}:${notification.request_id}` +} + +function projectLabel(project: { readonly name?: string; readonly worktree: string }) { + return project.name ?? (basename(project.worktree) || "project") +} + +function requireString(input: Record, key: string) { + const value = input[key] + if (typeof value !== "string" || value.trim().length === 0) return undefined + return value +} + +function knownSession(input: Record, known: ReadonlySet) { + const sessionID = requireString(input, "session_id") + if (!sessionID || !known.has(sessionID)) return undefined + return sessionID +} + +const decodeQuestionAnswers = Schema.decodeUnknownOption(Schema.Array(Question.Answer)) +const decodeFormAnswer = Schema.decodeUnknownOption(Form.Answer) + +function sessionParameters() { + return { + type: "object", + additionalProperties: false, + properties: { session_id: { type: "string" } }, + required: ["session_id"], + } +} + +function requestParameters() { + return { + type: "object", + additionalProperties: false, + properties: { session_id: { type: "string" }, request_id: { type: "string" } }, + required: ["session_id", "request_id"], + } +} + +function toolError(message: string, retryable = false) { + return { status: "error", message, retryable } +} diff --git a/packages/voice/src/pcm.ts b/packages/voice/src/pcm.ts new file mode 100644 index 0000000000..a9a685fa28 --- /dev/null +++ b/packages/voice/src/pcm.ts @@ -0,0 +1,19 @@ +export const PCM_SAMPLE_RATE = 24_000 +export const PCM_BYTES_PER_MS = (PCM_SAMPLE_RATE * 2) / 1_000 +export const PCM_METER_FRAME_MS = 33 + +export function pcmLevel(bytes: Buffer) { + const samples = Math.floor(bytes.length / 2) + if (samples === 0) return 0 + const stride = Math.max(1, Math.floor(samples / 1_200)) + let energy = 0 + let count = 0 + for (let index = 0; index < samples; index += stride) { + const sample = bytes.readInt16LE(index * 2) / 32_768 + energy += sample * sample + count += 1 + } + const rms = Math.sqrt(energy / count) + if (rms < 0.008) return 0 + return Math.min(1, Math.sqrt((rms - 0.008) / 0.18)) +} diff --git a/packages/voice/src/protocol-live.ts b/packages/voice/src/protocol-live.ts new file mode 100644 index 0000000000..977140f48c --- /dev/null +++ b/packages/voice/src/protocol-live.ts @@ -0,0 +1,332 @@ +import type { VoiceConnection, VoiceProtocol, VoiceProtocolEvent, VoiceProtocolOptions } from "./protocol" +import { decodeVoiceToolInput } from "./protocol" +import { Option, Schema } from "effect" + +const LiveItem = Schema.Struct({ + id: Schema.String, + type: Schema.String, + text: Schema.optional(Schema.String), + name: Schema.optional(Schema.String), + call_id: Schema.optional(Schema.String), + arguments: Schema.optional(Schema.String), +}) +const LiveTurn = Schema.Struct({ + id: Schema.String, + role: Schema.Literals(["user", "assistant"]), + transcript: Schema.String, +}) +const LiveEvent = Schema.Struct({ + type: Schema.String, + audio: Schema.optional(Schema.String), + delta: Schema.optional(Schema.String), + start_ms: Schema.optional(Schema.Number), + end_ms: Schema.optional(Schema.Number), + turn_id: Schema.optional(Schema.String), + turn: Schema.optional(LiveTurn), + item: Schema.optional(LiveItem), + error: Schema.optional( + Schema.Struct({ + code: Schema.optional(Schema.String), + message: Schema.optional(Schema.String), + }), + ), +}) +const decodeLiveEvent = Schema.decodeUnknownOption(Schema.fromJsonString(LiveEvent)) +type LiveTurn = Schema.Schema.Type +type LiveEvent = Schema.Schema.Type + +type ProjectedTurn = LiveTurn & { readonly displayID: string } + +export function createLiveEventProjector() { + const turns = new Map() + let input: { readonly id: string; readonly transcript: string } | undefined + let assistantTranscript = "" + const userTranscripts = new Map() + const startedTools = new Set() + + const startTool = (item: LiveEvent["item"], events: VoiceProtocolEvent[]) => { + if (item?.type !== "function_call" || !item.name || !item.call_id || startedTools.has(item.call_id)) return + startedTools.add(item.call_id) + events.push({ type: "tool.started", id: item.call_id, name: item.name }) + } + + const syncUser = (id: string, text: string, final: boolean, events: VoiceProtocolEvent[]) => { + text = text.trimStart() + const previous = userTranscripts.get(id) + if (previous?.text === text && previous.final === final) return + if (final) userTranscripts.delete(id) + else userTranscripts.set(id, { text, final }) + events.push({ type: "user.transcript", id, text, final }) + } + + // Deltas are emitted verbatim, matching the Realtime adapter: a fragment's leading space is + // the only record of the boundary between two assistant turns, so trimming it here would + // weld the last word of one turn onto the first word of the next. Stripping the leading + // edge of a rendered message is the consumer's job. + const syncAssistant = (transcript: string, events: VoiceProtocolEvent[]) => { + if (transcript.startsWith(assistantTranscript)) { + const delta = transcript.slice(assistantTranscript.length) + assistantTranscript = transcript + if (delta) events.push({ type: "assistant.transcript.delta", delta }) + return + } + if (assistantTranscript.startsWith(transcript)) { + assistantTranscript = transcript + events.push({ type: "assistant.transcript", text: transcript }) + return + } + assistantTranscript = transcript + events.push({ type: "assistant.transcript", text: transcript }) + } + + return (message: string) => { + const data = Option.getOrUndefined(decodeLiveEvent(message)) + if (!data) + return { + type: undefined, + events: [{ type: "error", message: "Received an invalid Live API event." }] satisfies VoiceProtocolEvent[], + } + + const events: VoiceProtocolEvent[] = data.type.endsWith(".delta") ? [] : [{ type: "debug", message: data.type }] + switch (data.type) { + case "session.started": + events.push({ type: "ready" }) + break + case "output_audio.delta": + if (data.audio && data.start_ms !== undefined && data.end_ms !== undefined) + events.push({ + type: "assistant.audio", + audio: Buffer.from(data.audio, "base64"), + timeline: { startMs: data.start_ms, endMs: data.end_ms }, + }) + break + case "input_transcript.added": + if (data.item?.type !== "input_transcript" || data.item.text === undefined) break + if (!input) { + input = { id: data.item.id, transcript: "" } + events.push({ type: "user.committed", id: input.id }) + } + input = { ...input, transcript: input.transcript + data.item.text } + syncUser(input.id, input.transcript, false, events) + break + case "output_transcript.added": + if (data.item?.type !== "output_transcript" || data.item.text === undefined) break + const delta = data.item.text + assistantTranscript += delta + if (delta) events.push({ type: "assistant.transcript.delta", delta }) + break + case "turn.created": { + const turn = data.turn + if (!turn) break + const displayID = turn.role === "user" ? (input?.id ?? turn.id) : turn.id + turns.set(turn.id, { ...turn, displayID }) + if (turn.role === "user") { + if (!input) events.push({ type: "user.committed", id: displayID }) + syncUser(displayID, turn.transcript, false, events) + break + } + syncAssistant(turn.transcript, events) + break + } + case "turn.delta": { + if (!data.turn_id || !data.delta) break + const turn = turns.get(data.turn_id) + if (!turn) break + const transcript = turn.transcript + data.delta + turns.set(data.turn_id, { ...turn, transcript }) + if (turn.role === "user") { + syncUser(turn.displayID, transcript, false, events) + break + } + syncAssistant(transcript, events) + break + } + case "turn.done": { + const turn = data.turn + if (!turn) break + const previous = turns.get(turn.id) + if (turn.role === "user") { + const displayID = previous?.displayID ?? input?.id ?? turn.id + if (!previous && !input) events.push({ type: "user.committed", id: displayID }) + syncUser(displayID, turn.transcript, true, events) + input = undefined + turns.delete(turn.id) + break + } + syncAssistant(turn.transcript, events) + assistantTranscript = "" + turns.delete(turn.id) + events.push({ type: "assistant.done", awaitingWork: false }) + break + } + case "response.output_item.added": + startTool(data.item, events) + break + case "response.output_item.done": { + if (data.item?.type !== "function_call" || !data.item.call_id) break + const name = data.item.name ?? "tool" + startTool(data.item, events) + if (!data.item.name || data.item.arguments === undefined) { + const output = { status: "error", message: "Malformed Live function call." } + events.push({ + type: "work.rejected", + request: { id: data.item.call_id, name, input: {} }, + output, + }) + startedTools.delete(data.item.call_id) + break + } + const input = Option.getOrUndefined(decodeVoiceToolInput(data.item.arguments)) + if (!input) { + const output = { status: "error", message: `Invalid arguments for tool ${data.item.name}.` } + events.push({ type: "error", message: `Received invalid arguments for Live tool ${data.item.name}.` }) + events.push({ + type: "work.rejected", + request: { id: data.item.call_id, name, input: {} }, + output, + }) + startedTools.delete(data.item.call_id) + break + } + events.push({ + type: "work.requested", + request: { id: data.item.call_id, name: data.item.name, input }, + }) + startedTools.delete(data.item.call_id) + break + } + case "error": + events.push({ type: "error", message: `${data.error?.code}: ${data.error?.message}` }) + } + return { type: data.type, events } + } +} + +export function createLiveProtocol(): VoiceProtocol { + return { + name: "live", + inputActivity: "local", + supportsTextInput: false, + connect: connectLive, + } +} + +function connectLive(options: VoiceProtocolOptions): VoiceConnection { + const ws = new WebSocket(`wss://api.openai.com/v1/live?model=${options.model}`, { + headers: { + Authorization: `Bearer ${options.apiKey}`, + "OpenAI-Alpha": "quicksilver=v2", + }, + } as unknown as string[]) + const closed = Promise.withResolvers() + let closeTimer: ReturnType | undefined + let notification: PromiseWithResolvers | undefined + let notificationTimer: ReturnType | undefined + const project = createLiveEventProjector() + + const settleNotification = (accepted: boolean) => { + if (!notification) return + if (notificationTimer) clearTimeout(notificationTimer) + notificationTimer = undefined + notification.resolve(accepted) + notification = undefined + } + + const send = (event: Record) => { + if (options.debug || event["type"] !== "input_audio.append") + options.trace?.("live.send", { + type: event["type"], + delegationID: event["delegation_item_id"], + }) + if (ws.readyState !== WebSocket.OPEN) return false + ws.send(JSON.stringify(event)) + return true + } + + ws.addEventListener("open", () => { + send({ + type: "session.update", + event_id: crypto.randomUUID(), + session: { + instructions: options.instructions, + audio: { output: { voice: options.voice } }, + delegation: { + type: "responses", + responses: { + model: options.delegationModel, + instructions: options.delegationInstructions, + tools: options.tools, + }, + }, + }, + }) + }) + ws.addEventListener("message", (event) => { + const result = project(String(event.data)) + if (options.debug || result.type !== "output_audio.delta") options.trace?.("live.receive", { type: result.type }) + if (result.type === "session.context.appended") settleNotification(true) + result.events.forEach(options.onEvent) + if (result.type === "session.closed") ws.close(1000) + }) + ws.addEventListener("close", (event) => { + if (closeTimer) clearTimeout(closeTimer) + settleNotification(false) + options.onEvent({ type: "closed", code: event.code }) + closed.resolve() + }) + + return { + appendAudio(audio) { + if (ws.bufferedAmount > 96_000) { + options.trace?.("live.audio.dropped", { bytes: audio.length, buffered: ws.bufferedAmount }) + return + } + send({ type: "input_audio.append", audio: audio.toString("base64") }) + }, + resolveWork(request, output) { + send({ + type: "delegation.function_call_output.create", + event_id: crypto.randomUUID(), + item: { type: "function_call_output", call_id: request.id, output: JSON.stringify(output) }, + }) + }, + notify(text) { + if (notification) return Promise.resolve(false) + notification = Promise.withResolvers() + if ( + !send({ + type: "session.context.append", + event_id: crypto.randomUUID(), + content: [{ type: "input_text", text: notificationText(text) }], + }) + ) { + settleNotification(false) + return Promise.resolve(false) + } + notificationTimer = setTimeout(() => settleNotification(false), 5_000) + return notification.promise + }, + interrupt() { + return false + }, + close(closeOptions) { + if (ws.readyState === WebSocket.CLOSED) return Promise.resolve() + if (closeOptions?.graceful === false || ws.readyState !== WebSocket.OPEN) { + ws.close(1000) + closeTimer = setTimeout(() => closed.resolve(), 5_000) + return closed.promise + } + send({ type: "session.close", event_id: crypto.randomUUID() }) + closeTimer = setTimeout(() => { + ws.close(1000) + closed.resolve() + }, 10_500) + return closed.promise + }, + } +} + +function notificationText(output: unknown) { + const text = String(output) + return text.length > 1_600 ? text.slice(0, 1_600) + "..." : text +} diff --git a/packages/voice/src/protocol-realtime.ts b/packages/voice/src/protocol-realtime.ts new file mode 100644 index 0000000000..d2031e1ca3 --- /dev/null +++ b/packages/voice/src/protocol-realtime.ts @@ -0,0 +1,280 @@ +import type { VoiceConnection, VoiceProtocol, VoiceProtocolEvent, VoiceProtocolOptions } from "./protocol" +import { decodeVoiceToolInput } from "./protocol" +import { Option, Schema } from "effect" + +const RealtimeItem = Schema.Struct({ + type: Schema.optional(Schema.String), + name: Schema.optional(Schema.String), + call_id: Schema.optional(Schema.String), + arguments: Schema.optional(Schema.String), +}) +const RealtimeFunctionCall = Schema.Struct({ + type: Schema.Literal("function_call"), + name: Schema.String, + call_id: Schema.String, + arguments: Schema.String, +}) +const RealtimeEvent = Schema.Struct({ + type: Schema.String, + delta: Schema.optional(Schema.String), + transcript: Schema.optional(Schema.String), + item_id: Schema.optional(Schema.String), + item: Schema.optional(RealtimeItem), + response: Schema.optional(Schema.Struct({ output: Schema.optional(Schema.Array(RealtimeItem)) })), + error: Schema.optional( + Schema.Struct({ + code: Schema.optional(Schema.String), + message: Schema.optional(Schema.String), + }), + ), +}) +const decodeRealtimeEvent = Schema.decodeUnknownOption(Schema.fromJsonString(RealtimeEvent)) +const decodeFunctionCall = Schema.decodeUnknownOption(RealtimeFunctionCall) +type RealtimeEvent = Schema.Schema.Type + +export function createRealtimeProtocol(): VoiceProtocol { + return { + name: "realtime", + inputActivity: "server", + supportsTextInput: true, + connect: connectRealtime, + } +} + +function connectRealtime(options: VoiceProtocolOptions): VoiceConnection { + const ws = new WebSocket(`wss://api.openai.com/v1/realtime?model=${options.model}`, { + headers: { Authorization: `Bearer ${options.apiKey}` }, + } as unknown as string[]) + const closed = Promise.withResolvers() + let closeTimer: ReturnType | undefined + let notification: PromiseWithResolvers | undefined + let notificationTimer: ReturnType | undefined + const pendingCalls = new Set() + const resolvedCalls = new Set() + const startedCalls = new Set() + let responseAwaitingWork = false + + const settleNotification = (accepted: boolean) => { + if (!notification) return + if (notificationTimer) clearTimeout(notificationTimer) + notificationTimer = undefined + notification.resolve(accepted) + notification = undefined + } + + const send = (event: Record) => { + if (options.debug || event["type"] !== "input_audio_buffer.append") + options.trace?.("realtime.send", { + type: event["type"], + callID: + event["item"] && typeof event["item"] === "object" && "call_id" in event["item"] + ? event["item"].call_id + : undefined, + }) + if (ws.readyState !== WebSocket.OPEN) return false + ws.send(JSON.stringify(event)) + return true + } + const createResponse = (text = false) => + send(text ? { type: "response.create", response: { output_modalities: ["text"] } } : { type: "response.create" }) + const resumeAfterWork = () => { + if (!responseAwaitingWork || pendingCalls.size > 0) return + responseAwaitingWork = false + resolvedCalls.clear() + createResponse() + } + const resolveFunctionCall = (id: string, output: unknown) => { + send({ + type: "conversation.item.create", + item: { type: "function_call_output", call_id: id, output: JSON.stringify(output) }, + }) + resolvedCalls.add(id) + pendingCalls.delete(id) + resumeAfterWork() + } + const startFunctionCall = (item: Schema.Schema.Type) => { + if (item.type !== "function_call" || !item.name || !item.call_id || startedCalls.has(item.call_id)) return + startedCalls.add(item.call_id) + options.onEvent({ type: "tool.started", id: item.call_id, name: item.name }) + } + const requestWork = (item: Schema.Schema.Type) => { + startFunctionCall(item) + const call = Option.getOrUndefined(decodeFunctionCall(item)) + if (!call) { + const output = { status: "error", message: "Malformed Realtime function call." } + options.onEvent({ type: "error", message: "Received a malformed Realtime function call." }) + if (item.call_id) { + options.onEvent({ + type: "work.rejected", + request: { id: item.call_id, name: item.name ?? "tool", input: {} }, + output, + }) + startedCalls.delete(item.call_id) + } + return + } + pendingCalls.add(call.call_id) + const input = Option.getOrUndefined(decodeVoiceToolInput(call.arguments)) + if (!input) { + const output = { status: "error", message: `Invalid arguments for tool ${call.name}.` } + options.onEvent({ type: "error", message: `Received invalid arguments for Realtime tool ${call.name}.` }) + options.onEvent({ + type: "work.rejected", + request: { id: call.call_id, name: call.name, input: {} }, + output, + }) + startedCalls.delete(call.call_id) + return + } + options.onEvent({ + type: "work.requested", + request: { id: call.call_id, name: call.name, input }, + }) + startedCalls.delete(call.call_id) + } + const finishResponse = (output: ReadonlyArray>) => { + const callIDs = output.flatMap((item) => (item.type === "function_call" && item.call_id ? [item.call_id] : [])) + callIDs.filter((id) => !resolvedCalls.has(id)).forEach((id) => pendingCalls.add(id)) + responseAwaitingWork = callIDs.length > 0 + options.onEvent({ type: "assistant.done", awaitingWork: responseAwaitingWork }) + resumeAfterWork() + } + + ws.addEventListener("open", () => { + send({ + type: "session.update", + session: { + type: "realtime", + instructions: options.instructions, + tools: options.tools, + tool_choice: "auto", + audio: { output: { voice: options.voice } }, + }, + }) + send({ + type: "session.update", + session: { type: "realtime", audio: { input: { transcription: { model: "whisper-1" } } } }, + }) + send({ + type: "session.update", + session: { + type: "realtime", + audio: { + input: { + turn_detection: { + type: "server_vad", + silence_duration_ms: 900, + interrupt_response: options.fullDuplex, + }, + }, + }, + }, + }) + }) + ws.addEventListener("message", (event) => { + const data = Option.getOrUndefined(decodeRealtimeEvent(String(event.data))) + if (!data) { + options.onEvent({ type: "error", message: "Received an invalid Realtime API event." }) + return + } + if (options.debug || data.type !== "response.output_audio.delta") + options.trace?.("realtime.receive", { type: data.type }) + if (data.type === "conversation.item.created") settleNotification(true) + onMessage(data, options.onEvent, startFunctionCall, requestWork, finishResponse) + }) + ws.addEventListener("close", (event) => { + if (closeTimer) clearTimeout(closeTimer) + settleNotification(false) + options.onEvent({ type: "closed", code: event.code }) + closed.resolve() + }) + + return { + appendAudio(audio) { + if (ws.bufferedAmount > 96_000) { + options.trace?.("realtime.audio.dropped", { bytes: audio.length, buffered: ws.bufferedAmount }) + return + } + send({ type: "input_audio_buffer.append", audio: audio.toString("base64") }) + }, + sendText(text) { + send({ + type: "conversation.item.create", + item: { type: "message", role: "user", content: [{ type: "input_text", text }] }, + }) + createResponse(true) + }, + resolveWork(request, output) { + resolveFunctionCall(request.id, output) + }, + notify(text) { + if (notification) return Promise.resolve(false) + notification = Promise.withResolvers() + const created = send({ + type: "conversation.item.create", + item: { type: "message", role: "user", content: [{ type: "input_text", text }] }, + }) + if (!created || !createResponse()) { + settleNotification(false) + return Promise.resolve(false) + } + notificationTimer = setTimeout(() => settleNotification(false), 5_000) + return notification.promise + }, + interrupt() { + send({ type: "response.cancel" }) + return true + }, + close() { + if (ws.readyState === WebSocket.CLOSED) return Promise.resolve() + ws.close(1000) + closeTimer ??= setTimeout(() => closed.resolve(), 5_000) + return closed.promise + }, + } +} + +function onMessage( + data: RealtimeEvent, + emit: (event: VoiceProtocolEvent) => void, + startFunctionCall: (item: Schema.Schema.Type) => void, + requestWork: (item: Schema.Schema.Type) => void, + finishResponse: (output: ReadonlyArray>) => void, +) { + if (!data.type.endsWith(".delta")) emit({ type: "debug", message: data.type }) + switch (data.type) { + case "session.created": + emit({ type: "ready" }) + return + case "response.output_text.delta": + case "response.output_audio_transcript.delta": + emit({ type: "assistant.transcript.delta", delta: data.delta ?? "" }) + return + case "response.done": + finishResponse(data.response?.output ?? []) + return + case "input_audio_buffer.speech_started": + emit({ type: "user.started" }) + return + case "input_audio_buffer.speech_stopped": + emit({ type: "user.stopped" }) + return + case "input_audio_buffer.committed": + emit({ type: "user.committed", id: data.item_id ?? "" }) + return + case "conversation.item.input_audio_transcription.completed": + emit({ type: "user.transcript", id: data.item_id ?? "", text: (data.transcript ?? "").trim(), final: true }) + return + case "response.output_audio.delta": + if (data.delta) emit({ type: "assistant.audio", audio: Buffer.from(data.delta, "base64") }) + return + case "response.output_item.added": + if (data.item?.type === "function_call") startFunctionCall(data.item) + return + case "response.output_item.done": + if (data.item?.type === "function_call") requestWork(data.item) + return + case "error": + emit({ type: "error", message: `${data.error?.code}: ${data.error?.message}` }) + } +} diff --git a/packages/voice/src/protocol.ts b/packages/voice/src/protocol.ts new file mode 100644 index 0000000000..b9314e516b --- /dev/null +++ b/packages/voice/src/protocol.ts @@ -0,0 +1,69 @@ +import { Schema } from "effect" + +export const decodeVoiceToolInput = Schema.decodeUnknownOption( + Schema.fromJsonString(Schema.Record(Schema.String, Schema.Unknown)), +) + +export type VoiceTool = { + readonly type: "function" + readonly name: string + readonly description: string + readonly parameters: unknown +} + +export type VoiceWorkRequest = { + readonly id: string + readonly name: string + readonly input: Record +} + +export type VoiceProtocolEvent = + | { readonly type: "ready" } + | { readonly type: "user.started" } + | { readonly type: "user.stopped" } + | { readonly type: "user.committed"; readonly id: string } + | { readonly type: "user.transcript"; readonly id: string; readonly text: string; readonly final: boolean } + | { + readonly type: "assistant.audio" + readonly audio: Buffer + readonly timeline?: { readonly startMs: number; readonly endMs: number } + } + | { readonly type: "assistant.transcript.delta"; readonly delta: string } + | { readonly type: "assistant.transcript"; readonly text: string } + | { readonly type: "assistant.done"; readonly awaitingWork: boolean } + | { readonly type: "tool.started"; readonly id: string; readonly name: string } + | { readonly type: "work.requested"; readonly request: VoiceWorkRequest } + | { readonly type: "work.rejected"; readonly request: VoiceWorkRequest; readonly output: unknown } + | { readonly type: "debug"; readonly message: string } + | { readonly type: "error"; readonly message: string } + | { readonly type: "closed"; readonly code: number } + +export type VoiceProtocolOptions = { + readonly apiKey: string + readonly model: string + readonly voice: string + readonly instructions: string + readonly delegationModel: string + readonly delegationInstructions: string + readonly tools: ReadonlyArray + readonly fullDuplex: boolean + readonly debug: boolean + readonly onEvent: (event: VoiceProtocolEvent) => void + readonly trace?: (event: string, data?: Record) => void +} + +export type VoiceConnection = { + appendAudio(audio: Buffer): void + sendText?(text: string): void + resolveWork(request: VoiceWorkRequest, output: unknown): void + notify(text: string): Promise + interrupt(): boolean + close(options?: { readonly graceful?: boolean }): Promise +} + +export type VoiceProtocol = { + readonly name: "live" | "realtime" + readonly inputActivity: "local" | "server" + readonly supportsTextInput: boolean + connect(options: VoiceProtocolOptions): VoiceConnection +} diff --git a/packages/voice/src/spike.ts b/packages/voice/src/spike.ts new file mode 100644 index 0000000000..c67ceecb39 --- /dev/null +++ b/packages/voice/src/spike.ts @@ -0,0 +1,682 @@ +#!/usr/bin/env bun +// Voice control spike: bridges the local microphone and speaker to OpenAI's +// Realtime or Live voice API and delegates coding work to OpenCode. +// +// Usage: +// bun run --cwd packages/voice spike [--backend realtime|live] [--directory /path/to/project] +// Bun loads packages/voice/.env automatically when the command runs from that package. +// +// Requires sox (`brew install sox`) for mic capture (`rec`) and playback (`play`). + +import { parseArgs } from "node:util" +import { OpenCode } from "@opencode-ai/client/promise" +import { Service } from "@opencode-ai/client/service" +import type { VoiceConnection, VoiceProtocolEvent, VoiceTool, VoiceWorkRequest } from "./protocol" +import { AudioJitterBuffer, type PlaybackChunk } from "./audio-jitter-buffer" +import { createOpenCodeBridge } from "./opencode" +import type { OpenCodeNotification } from "./opencode-notification" +import { pcmLevel, PCM_BYTES_PER_MS, PCM_METER_FRAME_MS, PCM_SAMPLE_RATE } from "./pcm" +import { createVoiceTrace } from "./trace" +import { initialVoiceState, transitionVoice, type VoiceCommand, type VoiceEvent } from "./voice-coordinator" + +const args = parseArgs({ + options: { + backend: { type: "string", default: "realtime" }, + server: { type: "string" }, + password: { type: "string" }, + directory: { type: "string", default: process.cwd() }, + model: { type: "string" }, + "delegation-model": { type: "string", default: "gpt-5.5" }, + voice: { type: "string", default: "marin" }, + provider: { type: "string", default: "openai" }, + "coding-model": { type: "string", default: "gpt-5.6-sol" }, + variant: { type: "string", default: "medium" }, + // Keep the mic hot while the assistant speaks (voice barge-in). Only + // usable with headphones: on speakers the mic hears the assistant and + // interrupts it with its own echo. Default is half-duplex gating. + duplex: { type: "boolean", default: false }, + // Enable Apple voice processing (echo cancellation) in the audio helper. + // Needed for full duplex on speakers; harmful with Bluetooth headsets, + // where it can bind the wrong capture device. + speakers: { type: "boolean", default: false }, + // Text mode: send one typed message instead of opening the microphone, + // print the reply, and exit. Useful for smoke-testing the tool loop. + text: { type: "string" }, + // Log every protocol event type as it arrives. + debug: { type: "boolean", default: false }, + "reduce-motion": { type: "boolean", default: false }, + }, +}).values +const trace = await createVoiceTrace() + +if (args.backend !== "realtime" && args.backend !== "live") { + console.error("--backend must be realtime or live") + process.exit(1) +} +const protocol = + args.backend === "live" + ? (await import("./protocol-live")).createLiveProtocol() + : (await import("./protocol-realtime")).createRealtimeProtocol() +const model = args.model ?? (protocol.name === "live" ? "gpt-live-1-boulder-alpha" : "gpt-realtime-2.1") +if (args.text && !protocol.supportsTextInput) { + console.error(`--text is not supported by the ${protocol.name} backend`) + process.exit(1) +} +const apiKey = (() => { + const value = process.env["OPENAI_API_KEY"] + if (value) return value + console.error("OPENAI_API_KEY is required. Add it to the gitignored packages/voice/.env or export it in the shell.") + process.exit(1) + return "" +})() +if (!args.text && !process.stdout.isTTY) { + console.error( + "The voice TUI requires direct terminal output; secret wrappers that pipe stdout cannot preserve resize.", + ) + console.error("Run directly: bun run --cwd packages/voice spike --backend " + protocol.name) + process.exit(1) +} + +const serverPassword = args.password ?? process.env["OPENCODE_PASSWORD"] ?? process.env["OPENCODE_SERVER_PASSWORD"] +const endpoint = args.server + ? { + url: args.server, + auth: serverPassword ? { type: "basic" as const, username: "opencode", password: serverPassword } : undefined, + } + : ((await Service.discover()) ?? (await Service.ensure({ command: ["opencode2", "serve", "--service"] }))) +const client = OpenCode.make({ + baseUrl: endpoint.url, + headers: Service.headers(endpoint), +}) +const health = await client.health.get().catch((error) => { + console.error(`Could not reach the OpenCode 2 server at ${endpoint.url}: ${error}`) + process.exit(1) +}) +// --------------------------------------------------------------------------- +// UI: OpenTUI in voice mode, plain console in --text mode. Created before the +// WebSocket so no await sits between socket creation and handler registration. +// --------------------------------------------------------------------------- + +const { createConsoleUI, createVoiceTUI } = await import("./ui") +const tuiActive = !args.text +const ui = tuiActive + ? await createVoiceTUI({ + onInterrupt: () => interrupt(), + onExit: () => shutdown(), + onCycleVoice: () => cycleVoice(), + onToggleMicrophone: () => toggleMicrophone(), + onToggleSpeaker: () => toggleSpeaker(), + reducedMotion: args["reduce-motion"], + }) + : createConsoleUI() +ui.setStatus({ server: endpoint.url, model: `${args["coding-model"]}:${args.variant}` }) +ui.meta(`opencode ${endpoint.url} (version ${health.version})`) +ui.meta(`project ${args.directory}`) +ui.meta(`trace ${trace.path}`) +trace.write("voice.started", { backend: protocol.name, model, directory: args.directory }) + +const voices = ["marin", "cedar", "coral", "sage", "ash", "ballad", "alloy", "verse"] +let voiceState = initialVoiceState(args.voice ?? "marin") +let connection: VoiceConnection | undefined +const opencode = await createOpenCodeBridge({ + client, + directory: args.directory, + model: { providerID: args.provider, id: args["coding-model"], variant: args.variant }, + notify: queueNotification, + trace: (event, data) => trace.write(event, data), + onSession: (sessionID) => { + ui.setStatus({ session: sessionID.slice(0, 12) }) + }, +}) +const toolDefinitions: ReadonlyArray = [ + ...opencode.definitions, + { + type: "function", + name: "set_voice", + description: "Change your speaking voice. Requires a brief reconnect and resets voice conversation memory.", + parameters: { + type: "object", + additionalProperties: false, + properties: { voice: { type: "string", enum: voices } }, + required: ["voice"], + }, + }, +] + +const baseInstructions = `You are the voice interface to OpenCode, a coding agent running on the user's machine. +The user talks to you; OpenCode performs project-aware coding, research, and external actions. You never write code yourself. + +Guidelines: +- Keep spoken replies to one or two sentences. +- Summarize coding-agent replies conversationally; do not read code, diffs, IDs, or paths aloud unless asked. +- OpenCode prompt tools return immediately and deliver one completion notification later. Stay conversational while work runs. +- Treat opencode.prompt.completed and opencode.prompt.blocked context as trusted client notifications, not user messages. +- Never claim delegated work succeeded until its result arrives. +- Explain failures briefly and offer one retry or an alternative.` + +const instructions = + protocol.name === "live" + ? `${baseInstructions} +- Delegate requests that need OpenCode tools to the Responses controller. +- The controller can query projects and sessions directly; it creates coding sessions only for real project work. +- Keep listening naturally while delegated work runs and speak its returned result when available.` + : `${baseInstructions} +- Use find_projects for explicit cross-project work. Never invent project IDs or expose filesystem paths. +- Use find_sessions to resolve references such as "the audio session". Search the current project first unless the user asks across projects. +- Use read_session to inspect a discovered Session directly. Do not prompt a Session merely to read its existing output. +- Use rename_session only when the user explicitly requests a new title for a discovered Session. Confirm the resulting title without reading its ID aloud. +- Before archive_session, name the discovered Session, explain that archiving hides it without stopping active work, and obtain explicit confirmation. Report the title, never its ID. +- Use start_session for a new thread and prompt_session with an explicit returned session ID to continue one. +- Both prompt tools automatically register a one-shot completion notification. Never wait or poll for completion. +- Before interrupt_session, state what will stop and obtain explicit confirmation. +- Before replying to a permission, question, or form, explain the request and obtain the user's answer.` + +const delegationInstructions = `You are the OpenCode controller behind a live voice assistant. +Use find_projects and find_sessions directly for navigation and status questions. +Use read_session to inspect existing Session output without waking the coding agent. +Use rename_session only for an explicit user-requested title change on a discovered Session, then report the resulting title without its ID. +Use archive_session only after the user explicitly confirms archiving the discovered Session. Archiving changes visibility but does not stop active work; report the title without its ID. +Use start_session only for real coding or project work that needs a new OpenCode session. +Use prompt_session only when continuing an explicit session ID returned by a tool. +Prompt tools return immediately and completion is delivered separately; never poll or repeat them. +Return concise factual text for the live assistant to summarize.` + +// --------------------------------------------------------------------------- +// Echo-cancelled full-duplex audio (Apple voice processing) +// --------------------------------------------------------------------------- + +// sox has no acoustic echo cancellation, so raw duplex on speakers feeds the +// assistant's voice back into the mic. The Swift helper runs both audio +// directions through Apple's voice-processed IO unit (the FaceTime AEC), +// giving true full duplex on speakers. Compiled on demand; sox is the +// fallback when swiftc is unavailable. +const aecBinary = await (async () => { + if (args.text || process.platform !== "darwin") return undefined + const source = Bun.fileURLToPath(new URL("./duplex-audio.swift", import.meta.url)) + const binary = Bun.fileURLToPath(new URL("../.build/duplex-audio", import.meta.url)) + if ((await Bun.file(binary).exists()) && Bun.file(binary).lastModified > Bun.file(source).lastModified) return binary + ui.meta("compiling echo-cancellation helper (first run only)...") + const { mkdir } = await import("node:fs/promises") + await mkdir(Bun.fileURLToPath(new URL("../.build", import.meta.url)), { recursive: true }) + const compile = Bun.spawn(["swiftc", "-O", source, "-o", binary], { stdout: "ignore", stderr: "pipe" }) + const [code, diagnostics] = await Promise.all([compile.exited, new Response(compile.stderr).text()]) + if (code === 0) return binary + ui.meta(diagnostics) + ui.meta("swiftc failed — falling back to sox audio") + return undefined +})() + +// Keep the mic hot during playback only with active AEC or an explicit +// headphone-mode opt-in. The Swift helper alone does not imply cancellation. +const fullDuplex = (aecBinary !== undefined && args.speakers) || args.duplex + +// Voice can't change once a session has produced audio, so switching voices +// reconnects the protocol (conversation context resets; the OpenCode session +// is untouched). +let microphoneMuted = false +let speakerMuted = false + +function queueNotification(notification: OpenCodeNotification) { + const promptID = "prompt_id" in notification ? notification.prompt_id : undefined + trace.write("notification.queued", { + type: notification.type, + promptID, + depth: voiceState.notifications.length + 1, + }) + dispatchVoice({ + type: "notification.queued", + notification: { promptID, text: JSON.stringify(notification) }, + }) +} + +function dispatchVoice(event: VoiceEvent) { + const transition = transitionVoice(voiceState, event) + if (transition.state === voiceState && transition.commands.length === 0) return + voiceState = transition.state + trace.write("voice.transition", { + input: event.type, + connection: voiceState.connection, + conversation: voiceState.conversation, + assistant: voiceState.assistant, + userSpeaking: voiceState.userSpeaking, + tools: voiceState.tools.size, + notifications: voiceState.notifications.length, + commands: transition.commands.map((command) => command.type), + }) + transition.commands.forEach(runVoiceCommand) +} + +function runVoiceCommand(command: VoiceCommand) { + if (command.type === "connection.reconnect") return reconnectVoice() + const active = connection + if (!active) return dispatchVoice({ type: "notification.failed", notification: command.notification }) + void active + .notify( + `OpenCode client notification. Announce this conversationally without reading IDs or raw JSON aloud:\n${command.notification.text}`, + ) + .then((accepted) => { + if (!accepted) return dispatchVoice({ type: "notification.failed", notification: command.notification }) + trace.write("notification.delivered", { + promptID: command.notification.promptID, + remaining: voiceState.notifications.length, + }) + if (command.notification.promptID) + void opencode + .delivered(command.notification.promptID) + .catch((error) => trace.write("notification.ack.failed", { error: String(error) })) + }) +} + +function setVoice(voice: string) { + ui.setStatus({ voice }) + ui.meta(`[voice] switching to ${voice}…`) + dispatchVoice({ type: "voice.selected", voice }) +} + +function reconnectVoice() { + if (shuttingDown) return + flushPlayback() + const previous = connection + connection = undefined + const reconnect = () => { + if (shuttingDown) return + connectProtocol() + } + if (!previous) return reconnect() + void previous.close({ graceful: false }).finally(reconnect) +} + +const cycleVoice = () => setVoice(voices[(voices.indexOf(voiceState.desiredVoice) + 1) % voices.length]) + +function toggleMicrophone() { + microphoneMuted = !microphoneMuted + if (userSpeechTimer) clearTimeout(userSpeechTimer) + userSpeechTimer = undefined + dispatchVoice({ type: "user.stopped" }) + ui.setStatus({ microphoneMuted }) + ui.userSpeaking(false) + if (microphoneMuted) ui.userReset() + ui.userAudioLevel(undefined) + ui.meta(`[microphone] ${microphoneMuted ? "muted" : "live"}`) +} + +function toggleSpeaker() { + speakerMuted = !speakerMuted + ui.setStatus({ speakerMuted }) + if (speakerMuted) flushPlayback(false) + ui.meta(`[speaker] ${speakerMuted ? "muted" : "live"}`) +} + +function interrupt() { + if (!assistantSpeaking() && voiceState.assistant === "idle") return + if (voiceState.assistant === "active" && !connection?.interrupt()) dispatchVoice({ type: "assistant.suppressed" }) + flushPlayback() + ui.meta("[interrupted]") +} + +let recorder: ReturnType | undefined +let player: ReturnType | undefined +let audio: ReturnType | undefined // AEC duplex helper (mic + speaker) + +// PCM16 mono 24kHz is the realtime API default; sox handles both directions. +const soxFormat = ["-q", "-t", "raw", "-r", String(PCM_SAMPLE_RATE), "-e", "signed-integer", "-b", "16", "-c", "1"] + +// Estimated wall-clock time when buffered speaker audio finishes playing. +// PCM16 mono at 24kHz is 48 bytes per millisecond. +let playbackEndsAt = 0 +const assistantSpeaking = () => Date.now() < playbackEndsAt +let playbackDoneTimer: ReturnType | undefined +const PLAYBACK_RELEASE_MS = 180 +const AUDIO_METER_BYTES = PCM_METER_FRAME_MS * PCM_BYTES_PER_MS +const playbackBuffer = new AudioJitterBuffer() +let outputTimelineEnd: number | undefined +let userFinalizedAt = 0 +let userSpeechTimer: ReturnType | undefined +let userDraftTimer: ReturnType | undefined +const USER_ACTIVITY_LEVEL = 0.2 +const USER_FINALIZED_COOLDOWN_MS = 500 +const AEC_PLAYBACK_FLOOR = 0.16 + +function observeUserAudio(level: number) { + ui.userAudioLevel(level) + if (protocol.inputActivity !== "local" || level < USER_ACTIVITY_LEVEL) return + if (Date.now() - userFinalizedAt < USER_FINALIZED_COOLDOWN_MS) return + if (assistantSpeaking() || voiceState.assistant === "active") { + if (fullDuplex && connection?.interrupt()) { + dispatchVoice({ type: "assistant.suppressed" }) + trace.write("assistant.barged", { level }) + flushPlayback() + } + return + } + if (userDraftTimer) clearTimeout(userDraftTimer) + userDraftTimer = undefined + if (!voiceState.userSpeaking) { + dispatchVoice({ type: "user.started" }) + ui.userSpeaking(true) + outputTimelineEnd = undefined + } + if (userSpeechTimer) clearTimeout(userSpeechTimer) + userSpeechTimer = setTimeout(() => { + userSpeechTimer = undefined + dispatchVoice({ type: "user.stopped" }) + ui.userSpeaking(false) + userDraftTimer = setTimeout(() => { + userDraftTimer = undefined + ui.userReset() + }, 1_200) + }, 500) +} + +let micStarted = false + +async function startMicrophone() { + if (micStarted) return // voice-switch reconnects reuse the running mic + micStarted = true + if (aecBinary) { + audio = Bun.spawn([aecBinary, ...(args.speakers ? ["--aec"] : [])], { + stdin: "pipe", + stdout: "pipe", + stderr: "pipe", + }) + void forwardHelperLogs(audio.stderr as ReadableStream) + ui.setStatus({ audio: args.speakers ? "duplex+aec" : args.duplex ? "duplex" : "half-duplex" }) + ui.meta(fullDuplex ? "mic live — talk any time, even over the assistant" : "mic live — pauses during playback") + for await (const chunk of audio.stdout as ReadableStream) { + if (!connection || microphoneMuted) continue + if (!fullDuplex && Date.now() < playbackEndsAt + 300) continue + const bytes = Buffer.from(chunk) + const level = pcmLevel(bytes) + observeUserAudio(level) + connection.appendAudio( + args.speakers && (assistantSpeaking() || voiceState.assistant === "active") && level < AEC_PLAYBACK_FLOOR + ? Buffer.alloc(bytes.length) + : bytes, + ) + } + return + } + recorder = Bun.spawn(["rec", ...soxFormat, "-"], { stdout: "pipe", stderr: "ignore" }) + ui.setStatus({ audio: args.duplex ? "duplex (sox)" : "half-duplex (sox)" }) + ui.meta("mic live — start talking") + if (!args.duplex) ui.meta("mic mutes while the assistant speaks; press Esc to interrupt") + for await (const chunk of recorder.stdout as ReadableStream) { + if (!connection || microphoneMuted) continue + // Half-duplex: drop mic audio while the assistant is audible (plus a + // short tail) so speaker echo can't barge-in against itself. + if (!args.duplex && Date.now() < playbackEndsAt + 300) continue + const bytes = Buffer.from(chunk) + observeUserAudio(pcmLevel(bytes)) + connection.appendAudio(bytes) + } +} + +async function forwardHelperLogs(stream: ReadableStream) { + const decoder = new TextDecoder() + let remainder = "" + for await (const chunk of stream) { + const lines = (remainder + decoder.decode(chunk, { stream: true })).split("\n") + remainder = lines.pop() ?? "" + for (const line of lines) { + if (!line.trim()) continue + ui.meta(line.trim()) + trace.write("audio.helper", { message: line.trim() }) + } + } + remainder += decoder.decode() + if (!remainder.trim()) return + ui.meta(remainder.trim()) + trace.write("audio.helper", { message: remainder.trim() }) +} + +function playAudio(bytes: Buffer, timeline?: { readonly startMs: number; readonly endMs: number }) { + if (speakerMuted) return + if (playbackDoneTimer) clearTimeout(playbackDoneTimer) + playbackDoneTimer = undefined + const timelineGap = + timeline && outputTimelineEnd !== undefined ? Math.max(0, timeline.startMs - outputTimelineEnd) : 0 + const gapMs = timelineGap < 2_000 ? timelineGap : 0 + if (args.debug) + trace.write("audio.output", { + bytes: bytes.length, + durationMs: bytes.length / PCM_BYTES_PER_MS, + level: pcmLevel(bytes), + timelineStart: timeline?.startMs, + timelineEnd: timeline?.endMs, + timelineGap, + gapMs, + }) + outputTimelineEnd = timeline?.endMs + playbackBuffer.push({ bytes, gapMs }).forEach(writeAudio) +} + +function writeAudio(chunk: PlaybackChunk) { + if (!audio) { + player ??= Bun.spawn(["play", ...soxFormat, "-"], { stdin: "pipe", stderr: "ignore" }) + if (args.debug) void player.exited.then((code) => ui.meta(`[debug] play exited (${code})`)) + } + const stdin = (audio ?? player)!.stdin as import("bun").FileSink + if (chunk.gapMs > 0) { + ui.assistantAudio(0, chunk.gapMs) + void stdin.write(Buffer.alloc(Math.round(chunk.gapMs * PCM_BYTES_PER_MS))) + } + for (let offset = 0; offset < chunk.bytes.length; offset += AUDIO_METER_BYTES) { + const window = chunk.bytes.subarray(offset, offset + AUDIO_METER_BYTES) + ui.assistantAudio(pcmLevel(window), window.length / PCM_BYTES_PER_MS) + } + void stdin.write(chunk.bytes) + void stdin.flush() + playbackEndsAt = Math.max(playbackEndsAt, Date.now()) + chunk.gapMs + chunk.bytes.length / PCM_BYTES_PER_MS +} + +function finishPlayback() { + playbackBuffer.finish().forEach(writeAudio) + if (playbackDoneTimer) clearTimeout(playbackDoneTimer) + playbackDoneTimer = setTimeout( + () => { + playbackDoneTimer = undefined + playbackBuffer.reset() + outputTimelineEnd = undefined + ui.assistantDone() + }, + Math.max(0, playbackEndsAt - Date.now()) + PLAYBACK_RELEASE_MS, + ) +} + +function flushPlayback(finishAssistant = true) { + // The AEC helper flushes its queued speaker audio on SIGUSR1 and keeps running. + if (audio) process.kill(audio.pid, "SIGUSR1") + player?.kill() + player = undefined + playbackEndsAt = 0 + playbackBuffer.reset() + if (playbackDoneTimer) clearTimeout(playbackDoneTimer) + playbackDoneTimer = undefined + outputTimelineEnd = undefined + if (finishAssistant) ui.assistantDone() + else ui.assistantPlaybackStopped() +} + +async function handleWork(source: VoiceConnection, request: VoiceWorkRequest) { + trace.write("work.started", { id: request.id, name: request.name }) + const output = await executeTool(request.name, request.input).catch((error) => toolError(String(error), true)) + trace.write("work.resolved", { id: request.id, name: request.name }) + ui.toolDone(request.id, output) + source.resolveWork(request, output) +} + +function queueWork(source: VoiceConnection, request: VoiceWorkRequest) { + void handleWork(source, request) + .catch((error) => ui.meta(`[work error] ${String(error)}`)) + .finally(() => dispatchVoice({ type: "tool.finished", id: request.id })) +} + +async function executeTool(name: string, input: Record) { + if (name !== "set_voice") return opencode.execute(name, input) + const voice = input["voice"] + if (typeof voice !== "string" || !voices.includes(voice)) + return toolError(`Voice must be one of: ${voices.join(", ")}.`) + setTimeout(() => setVoice(voice), 1_000) + return { status: "switching", voice, note: `${protocol.name} conversation memory resets during reconnect.` } +} + +function connectProtocol() { + let next: VoiceConnection + next = protocol.connect({ + apiKey, + model, + voice: voiceState.desiredVoice, + instructions, + delegationModel: args["delegation-model"], + delegationInstructions, + tools: toolDefinitions, + fullDuplex, + debug: args.debug, + onEvent: (event) => onProtocolEvent(next, event), + trace: (event, data) => trace.write(event, data), + }) + dispatchVoice({ type: "connection.connecting" }) + connection = next +} + +function onProtocolEvent(source: VoiceConnection, event: VoiceProtocolEvent) { + if (source !== connection) return + if (args.debug || event.type !== "assistant.audio") + trace.write("protocol.event", { + type: event.type, + conversation: voiceState.conversation, + assistant: voiceState.assistant, + userSpeaking: voiceState.userSpeaking, + pendingWork: voiceState.tools.size, + }) + switch (event.type) { + case "ready": + dispatchVoice({ type: "connection.ready" }) + ui.meta(`connected to ${protocol.name} ${model} (voice: ${voiceState.desiredVoice})`) + ui.setStatus({ voice: voiceState.desiredVoice, microphoneMuted, speakerMuted }) + if (!args.text) { + void startMicrophone() + return + } + ui.userTranscript("typed", args.text) + dispatchVoice({ type: "user.committed" }) + source.sendText?.(args.text) + return + case "user.started": + if (fullDuplex && voiceState.assistant === "active") dispatchVoice({ type: "assistant.suppressed" }) + dispatchVoice({ type: "user.started" }) + ui.userSpeaking(true) + if (fullDuplex) flushPlayback() + return + case "user.stopped": + dispatchVoice({ type: "user.stopped" }) + ui.userSpeaking(false) + return + case "user.committed": + if (userDraftTimer) clearTimeout(userDraftTimer) + userDraftTimer = undefined + dispatchVoice({ type: "user.committed" }) + ui.userCommitted(event.id) + return + case "user.transcript": + if (userDraftTimer) clearTimeout(userDraftTimer) + userDraftTimer = undefined + if (event.final) { + if (userSpeechTimer) clearTimeout(userSpeechTimer) + userSpeechTimer = undefined + dispatchVoice({ type: "user.stopped" }) + userFinalizedAt = Date.now() + ui.userSpeaking(false) + } + ui.userTranscript(event.id, event.text, event.final) + return + case "assistant.audio": + if (voiceState.assistant === "suppressed") return + dispatchVoice({ type: "assistant.started" }) + playAudio(event.audio, event.timeline) + return + case "assistant.transcript.delta": + if (voiceState.assistant === "suppressed") return + dispatchVoice({ type: "assistant.started" }) + ui.assistantDelta(event.delta) + return + case "assistant.transcript": + if (voiceState.assistant === "suppressed") return + dispatchVoice({ type: "assistant.started" }) + ui.assistantTranscript(event.text) + return + case "assistant.done": + dispatchVoice({ type: "assistant.done", awaitingWork: event.awaitingWork }) + if (args.text) ui.assistantDone() + else finishPlayback() + if (args.text && !event.awaitingWork && voiceState.tools.size === 0) shutdown() + return + case "tool.started": + dispatchVoice({ type: "tool.started", id: event.id }) + ui.toolStart(event.id, event.name, {}) + return + case "work.requested": + queueWork(source, event.request) + return + case "work.rejected": + dispatchVoice({ type: "tool.started", id: event.request.id }) + ui.toolStart(event.request.id, event.request.name, event.request.input) + ui.toolDone(event.request.id, event.output) + source.resolveWork(event.request, event.output) + dispatchVoice({ type: "tool.finished", id: event.request.id }) + return + case "debug": + if (args.debug) ui.meta(`[debug] ${event.message}`) + return + case "error": + ui.meta(`[${protocol.name} error] ${event.message}`) + return + case "closed": + dispatchVoice({ type: "connection.closed" }) + ui.meta(`${protocol.name} connection closed (${event.code})`) + shutdown() + } +} + +let shuttingDown = false +let shutdownFinished = false + +function finishShutdown() { + if (shutdownFinished) return + shutdownFinished = true + ui.close() + process.exit(0) +} + +function shutdown() { + if (shuttingDown) return + shuttingDown = true + recorder?.kill() + player?.kill() + audio?.kill() + if (userSpeechTimer) clearTimeout(userSpeechTimer) + if (userDraftTimer) clearTimeout(userDraftTimer) + if (playbackDoneTimer) clearTimeout(playbackDoneTimer) + const active = connection + connection = undefined + dispatchVoice({ type: "connection.closed" }) + trace.write("voice.shutdown") + const closeOpenCode = opencode.close().then(() => trace.write("voice.shutdown.opencode")) + const closeProtocol = active?.close().then(() => trace.write("voice.shutdown.protocol")) + const cleanup = Promise.allSettled([closeOpenCode, ...(closeProtocol ? [closeProtocol] : [])]) + void Promise.race([cleanup, Bun.sleep(5_000).then(() => trace.write("voice.shutdown.timeout"))]) + .then(() => trace.close()) + .finally(finishShutdown) +} + +function toolError(message: string, retryable = false) { + return { status: "error", message, retryable } +} + +// A surviving process keeps the microphone hot and the OpenAI meter running, +// so every terminal-death signal must tear it down. +process.on("SIGINT", shutdown) +process.on("SIGHUP", shutdown) +process.on("SIGTERM", shutdown) + +connectProtocol() diff --git a/packages/voice/src/trace.ts b/packages/voice/src/trace.ts new file mode 100644 index 0000000000..4d5d0ae1bf --- /dev/null +++ b/packages/voice/src/trace.ts @@ -0,0 +1,32 @@ +import { mkdir, readdir, unlink } from "node:fs/promises" +import { tmpdir } from "node:os" +import { join } from "node:path" + +export async function createVoiceTrace() { + const directory = join(tmpdir(), "opencode-voice") + await mkdir(directory, { recursive: true }) + const stale = (await readdir(directory)) + .filter((name) => name.endsWith(".jsonl")) + .sort() + .slice(0, -20) + await Promise.allSettled(stale.map((name) => unlink(join(directory, name)))) + const path = join(directory, `${new Date().toISOString().replaceAll(":", "-")}-${process.pid}.jsonl`) + const writer = Bun.file(path).writer() + const flush = setInterval(() => void writer.flush(), 250) + let bytes = 0 + flush.unref() + + return { + path, + write(event: string, data: Record = {}) { + if (bytes >= 10_000_000) return + const line = `${JSON.stringify({ time: Date.now(), event, ...data })}\n` + bytes += Buffer.byteLength(line) + void writer.write(line) + }, + async close() { + clearInterval(flush) + await writer.end() + }, + } +} diff --git a/packages/voice/src/ui-model.ts b/packages/voice/src/ui-model.ts new file mode 100644 index 0000000000..8f9e6f3d9a --- /dev/null +++ b/packages/voice/src/ui-model.ts @@ -0,0 +1,302 @@ +import { REVEAL_WORD_LIMIT, scheduleTextReveal, type TextReveal } from "./animation" + +export type Message = + | { + readonly key: string + readonly kind: "user" + readonly itemID: string + readonly text?: string + readonly transcribing: boolean + readonly reveals: ReadonlyArray + } + | { + readonly key: string + readonly kind: "assistant" + readonly text: string + readonly streaming: boolean + readonly reveals: ReadonlyArray + } + | { + readonly key: string + readonly kind: "tool" + readonly callID: string + readonly name: string + readonly input: unknown + readonly output?: unknown + } + | { readonly key: string; readonly kind: "meta"; readonly text: string } + +type NewMessage = Message extends infer Item ? (Item extends Message ? Omit : never) : never + +export type VoiceViewState = { + readonly messages: ReadonlyArray + readonly messageSequence: number + readonly activeUserID?: string + readonly userSequence: number + readonly revealAt: number + readonly revealAnimationEndsAt: number +} + +export type VoiceViewEvent = + | { readonly type: "meta"; readonly text: string } + | { readonly type: "user.started" } + | { readonly type: "user.reset" } + | { readonly type: "user.committed"; readonly itemID: string } + | { + readonly type: "user.transcript" + readonly itemID: string + readonly text: string + readonly final: boolean + readonly now: number + readonly animate: boolean + } + | { readonly type: "assistant.delta"; readonly text: string; readonly now: number; readonly animate: boolean } + | { readonly type: "assistant.transcript"; readonly text: string; readonly now: number; readonly animate: boolean } + | { readonly type: "assistant.done" } + | { readonly type: "tool.started"; readonly callID: string; readonly name: string; readonly input: unknown } + | { readonly type: "tool.done"; readonly callID: string; readonly output: unknown } + | { readonly type: "reveals.completed" } + +const MESSAGE_LIMIT = 200 + +export function initialVoiceView(): VoiceViewState { + return { messages: [], messageSequence: 0, userSequence: 0, revealAt: 0, revealAnimationEndsAt: 0 } +} + +export function transitionVoiceView(state: VoiceViewState, event: VoiceViewEvent): VoiceViewState { + switch (event.type) { + case "meta": + return append(state, { kind: "meta", text: event.text }) + case "user.started": { + if (state.activeUserID) return state + const itemID = `local-user-${state.userSequence + 1}` + return append( + { ...state, activeUserID: itemID, userSequence: state.userSequence + 1 }, + { kind: "user", itemID, transcribing: true, reveals: [] }, + ) + } + case "user.reset": { + if (!state.activeUserID) return state + return { + ...state, + activeUserID: undefined, + messages: mergeAssistantRows( + state.messages.flatMap((message) => { + if (message.kind !== "user" || message.itemID !== state.activeUserID) return [message] + if (!message.text) return [] + return [{ ...message, transcribing: false }] + }), + ), + } + } + case "user.committed": { + const index = state.activeUserID + ? state.messages.findIndex((message) => message.kind === "user" && message.itemID === state.activeUserID) + : -1 + const current = { + ...state, + activeUserID: undefined, + messages: state.messages.map((message, currentIndex) => { + if (message.kind === "assistant" && message.streaming) return { ...message, streaming: false } + if (currentIndex === index && message.kind === "user") return { ...message, itemID: event.itemID } + return message + }), + } + if (index !== -1) return current + return append(current, { kind: "user", itemID: event.itemID, transcribing: true, reveals: [] }) + } + case "user.transcript": { + const index = state.messages.findIndex((message) => message.kind === "user" && message.itemID === event.itemID) + if (index === -1) { + const revealed = reveal(state, "", event.text, event.now, event.animate) + return append(revealed.state, { + kind: "user", + itemID: event.itemID, + text: event.text, + transcribing: !event.final, + reveals: revealed.reveals, + }) + } + const previous = state.messages[index] + if (previous.kind !== "user") return state + const previousText = previous.text ?? "" + if (previousText === event.text && previous.transcribing === !event.final) return state + const appended = event.text.startsWith(previousText) + const revealed = reveal( + state, + appended ? previousText : "", + appended ? event.text.slice(previousText.length) : event.text, + event.now, + event.animate, + ) + return { + ...revealed.state, + messages: state.messages.map((message, currentIndex) => + currentIndex === index && message.kind === "user" + ? { + ...message, + text: event.text, + transcribing: !event.final, + reveals: appended + ? [...message.reveals, ...revealed.reveals].slice(-REVEAL_WORD_LIMIT) + : revealed.reveals, + } + : message, + ), + } + } + case "assistant.delta": { + const streaming = state.messages.findLastIndex((message) => message.kind === "assistant" && message.streaming) + const lastIndex = state.messages.length - 1 + const index = streaming === -1 && state.messages[lastIndex]?.kind === "assistant" ? lastIndex : streaming + const message = state.messages[index] + if (message?.kind !== "assistant") { + const text = event.text.trimStart() + const revealed = reveal(state, "", text, event.now, event.animate) + return append( + { + ...revealed.state, + messages: state.messages.map((message) => + message.kind === "assistant" && message.streaming ? { ...message, streaming: false } : message, + ), + }, + { kind: "assistant", text, streaming: true, reveals: revealed.reveals }, + ) + } + const joined = + streaming === -1 + ? joinAssistantText(message.text, event.text) + : { text: message.text + event.text, previous: message.text, appended: event.text } + const revealed = reveal(state, joined.previous, joined.appended, event.now, event.animate) + return { + ...revealed.state, + messages: state.messages.map((message, currentIndex) => + currentIndex === index && message.kind === "assistant" + ? { + ...message, + text: joined.text, + streaming: true, + reveals: [...message.reveals, ...revealed.reveals].slice(-REVEAL_WORD_LIMIT), + } + : message, + ), + } + } + case "assistant.transcript": { + const text = event.text.trimStart() + const index = state.messages.findLastIndex((message) => message.kind === "assistant" && message.streaming) + if (index === -1) { + const revealed = reveal(state, "", text, event.now, event.animate) + return append(revealed.state, { + kind: "assistant", + text, + streaming: true, + reveals: revealed.reveals, + }) + } + const previous = state.messages[index] + if (previous.kind !== "assistant" || previous.text === text) return state + const appended = text.startsWith(previous.text) + const revealed = reveal( + state, + appended ? previous.text : "", + appended ? text.slice(previous.text.length) : text, + event.now, + event.animate, + ) + return { + ...revealed.state, + messages: state.messages.map((message, currentIndex) => + currentIndex === index && message.kind === "assistant" + ? { + ...message, + text, + reveals: appended + ? [...message.reveals, ...revealed.reveals].slice(-REVEAL_WORD_LIMIT) + : revealed.reveals, + } + : message, + ), + } + } + case "assistant.done": { + const index = state.messages.findLastIndex((message) => message.kind === "assistant" && message.streaming) + if (index === -1) return state + return { + ...state, + messages: state.messages.map((message, currentIndex) => + currentIndex === index && message.kind === "assistant" ? { ...message, streaming: false } : message, + ), + } + } + case "tool.started": + return append(state, { + kind: "tool", + callID: event.callID, + name: event.name, + input: event.input, + }) + case "tool.done": + return { + ...state, + messages: state.messages.map((message) => + message.kind === "tool" && message.callID === event.callID ? { ...message, output: event.output } : message, + ), + } + case "reveals.completed": + return { + ...state, + revealAnimationEndsAt: 0, + messages: state.messages.map((message) => + "reveals" in message && message.reveals.length > 0 ? { ...message, reveals: [] } : message, + ), + } + } +} + +function append(state: VoiceViewState, message: NewMessage): VoiceViewState { + const messageSequence = state.messageSequence + 1 + const next = { ...message, key: `message-${messageSequence}` } + return { ...state, messageSequence, messages: [...state.messages, next].slice(-MESSAGE_LIMIT) } +} + +function reveal(state: VoiceViewState, previous: string, delta: string, now: number, animate: boolean) { + if (!animate) return { state, reveals: new Array() } + const scheduled = scheduleTextReveal(previous, delta, now, state.revealAt) + return { + state: { + ...state, + revealAt: scheduled.nextRevealAt, + revealAnimationEndsAt: Math.max(state.revealAnimationEndsAt, scheduled.animationEndsAt), + }, + reveals: scheduled.reveals, + } +} + +function mergeAssistantRows(messages: ReadonlyArray) { + return messages.reduce((result, message) => { + const previous = result.at(-1) + if (previous?.kind !== "assistant" || message.kind !== "assistant") return [...result, message] + return [ + ...result.slice(0, -1), + { + key: previous.key, + kind: "assistant", + text: joinAssistantText(previous.text, message.text).text, + streaming: previous.streaming || message.streaming, + reveals: [], + }, + ] + }, []) +} + +export function joinAssistantText(left: string, right: string) { + if (left.trim() === "") return { text: right, previous: "", appended: right } + if (right.trim() === "") return { text: left, previous: left, appended: "" } + const head = left.trimEnd() + const tail = right.trimStart() + if ((left.slice(head.length) + right.slice(0, right.length - tail.length)).includes("\n")) + return { text: left + right, previous: left, appended: right } + const separator = /^[.,;:!?…%)\]}]/.test(tail) || /[([{]$/.test(head) ? "" : " " + return { text: head + separator + tail, previous: head + separator, appended: tail } +} diff --git a/packages/voice/src/ui.tsx b/packages/voice/src/ui.tsx new file mode 100644 index 0000000000..7ea8572533 --- /dev/null +++ b/packages/voice/src/ui.tsx @@ -0,0 +1,657 @@ +/** @jsxImportSource @opentui/solid */ +// Terminal UI for the voice spike. The TUI keeps conversation order stable +// even though realtime events arrive out of order: a user row is inserted the +// moment the audio buffer commits (before the assistant starts replying) and +// its transcript is filled in when Whisper finishes. +import { createCliRenderer, RGBA } from "@opentui/core" +import { render, useKeyboard } from "@opentui/solid" +import "opentui-spinner/solid" +import { createSignal, For } from "solid-js" +import { createStore, reconcile } from "solid-js/store" +import { springOpacity, transcriptionPulse, type TextReveal } from "./animation" +import { PCM_METER_FRAME_MS } from "./pcm" +import { initialVoiceView, transitionVoiceView, type Message, type VoiceViewEvent } from "./ui-model" + +export { joinAssistantText } from "./ui-model" + +export type VoiceStatus = { + server?: string + session?: string + audio?: string + microphoneMuted?: boolean + speakerMuted?: boolean + voice?: string + model?: string + project?: string +} + +export type VoiceUI = { + meta(text: string): void + userSpeaking(active: boolean): void + userReset(): void + userAudioLevel(level?: number): void + userCommitted(itemID: string): void + userTranscript(itemID: string, text: string, final?: boolean): void + assistantAudio(level: number, durationMs: number): void + assistantPlaybackStopped(): void + assistantDelta(text: string): void + assistantTranscript(text: string): void + assistantDone(): void + toolStart(callID: string, name: string, input: unknown): void + toolDone(callID: string, output: unknown): void + setStatus(patch: VoiceStatus): void + close(): void +} + +const truncate = (text: string, max: number) => (text.length > max ? text.slice(0, max) + "…" : text) + +const displayJson = (value: unknown) => + truncate( + (() => { + try { + return ( + JSON.stringify( + value, + (_, item) => + typeof item === "string" ? truncate(item, 500) : typeof item === "bigint" ? String(item) : item, + 2, + ) ?? "null" + ) + } catch { + return "[unserializable value]" + } + })(), + 8_000, + ) + +function toolSummary(name: string, output?: unknown) { + if (output === undefined) return "running" + if (Array.isArray(output)) return `${output.length} result${output.length === 1 ? "" : "s"}` + if (typeof output === "string") return truncate(output, 160) + if (typeof output === "number" || typeof output === "boolean" || typeof output === "bigint") return String(output) + if (output === null || typeof output !== "object") return "completed" + if ("status" in output && typeof output.status === "string") + return output.status === "started" && "notification" in output && output.notification === "registered" + ? "started · notification registered" + : output.status + const fields = Object.entries(output) + .filter( + ([key, item]) => + !key.endsWith("_id") && key !== "notification" && ["string", "number", "boolean"].includes(typeof item), + ) + .slice(0, 2) + .map(([key, item]) => `${key}: ${String(item)}`) + return fields.join(" · ") || "completed" +} + +// --------------------------------------------------------------------------- +// Console fallback (--text mode, non-TTY) +// --------------------------------------------------------------------------- + +export function createConsoleUI(): VoiceUI { + const tty = process.stdout.isTTY + const dim = (text: string) => (tty ? `\x1b[2m${text}\x1b[0m` : text) + const cyan = (text: string) => (tty ? `\x1b[1;36m${text}\x1b[0m` : text) + const green = (text: string) => (tty ? `\x1b[1;32m${text}\x1b[0m` : text) + + let streaming = false + const tools = new Map() + const line = (text: string) => { + if (streaming) { + process.stdout.write("\n") + streaming = false + } + console.log(text) + } + + return { + meta: (text) => line(dim(` ${text}`)), + userSpeaking: () => {}, + userReset: () => {}, + userAudioLevel: () => {}, + userCommitted: () => {}, + userTranscript: (_, text) => line(cyan("● you ") + text), + assistantAudio: () => {}, + assistantPlaybackStopped: () => {}, + assistantDelta: (text) => { + if (!streaming) { + process.stdout.write(green("● assistant ")) + streaming = true + } + process.stdout.write(text) + }, + assistantTranscript: (text) => line(green("● assistant ") + text), + assistantDone: () => { + if (streaming) process.stdout.write("\n") + streaming = false + }, + toolStart: (callID, name) => { + tools.set(callID, name) + line(dim(` ◌ ${name} running`)) + }, + toolDone: (callID, output) => { + const name = tools.get(callID) ?? "tool" + tools.delete(callID) + line(dim(` ✓ ${name} ${toolSummary(name, output)}`)) + }, + setStatus: () => {}, + close: () => {}, + } +} + +// --------------------------------------------------------------------------- +// OpenTUI +// --------------------------------------------------------------------------- + +const theme = { + text: "#c0caf5", + muted: "#565f89", + you: "#7dcfff", + assistant: "#9ece6a", + key: "#7aa2f7", + string: "#9ece6a", + number: "#e0af68", + literal: "#bb9af7", +} + +const SPINNER_FRAMES = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏"] +const colors = { + text: RGBA.fromHex(theme.text), + you: RGBA.fromHex(theme.you), + assistant: RGBA.fromHex(theme.assistant), +} +const AUDIO_LEVEL_STALE_MS = 250 +const TOOL_DISPLAY_DELAY_MS = 160 +const meterGlyphs = ["▁", "▂", "▃", "▄", "▅", "▆", "▇", "█"] + +function fade(color: RGBA, opacity: number) { + return RGBA.fromValues(color.r, color.g, color.b, color.a * opacity) +} + +function meter(level: number, phase: number) { + return [0, 1, 2, 3] + .map((index) => { + const value = Math.max(0, Math.min(1, level * (0.72 + Math.sin(phase + index * 1.4) * 0.28))) + return meterGlyphs[Math.round(value * (meterGlyphs.length - 1))] + }) + .join("") +} + +// Tiny JSON tokenizer for syntax-highlighted tool results. +function jsonTokens(json: string) { + const tokens: Array<{ text: string; color: string }> = [] + const pattern = /("(?:[^"\\]|\\.)*")(\s*:)?|(-?\d+\.?\d*(?:[eE][+-]?\d+)?)|(true|false|null)|([{}[\],:]+|\s+)/g + for (const match of json.matchAll(pattern)) { + if (match[1] !== undefined) { + tokens.push({ text: match[1], color: match[2] ? theme.key : theme.string }) + if (match[2]) tokens.push({ text: match[2], color: theme.muted }) + continue + } + if (match[3] !== undefined) { + tokens.push({ text: match[3], color: theme.number }) + continue + } + if (match[4] !== undefined) { + tokens.push({ text: match[4], color: theme.literal }) + continue + } + tokens.push({ text: match[5] ?? "", color: theme.muted }) + } + return tokens +} + +function RevealedText(props: { + text: string + reveals: ReadonlyArray + now: () => number + opacity: () => number +}) { + return ( + <> + + {props.text.slice(0, props.reveals[0]?.offset ?? props.text.length)} + + + {(reveal, index) => ( + + {props.text.slice(reveal.offset, props.reveals[index() + 1]?.offset ?? props.text.length)} + + )} + + + ) +} + +// Message transitions preserve unchanged object identities. A changed message +// gets a new object, so keyed replaces that row and this branch reruns. +function MessageRow(props: { + message: Message + details: boolean + assistantLevel: () => number + userLevel: () => number + now: () => number + animate: boolean + microphoneMuted: boolean +}) { + const message = props.message + if (message.kind === "user") { + const transcribing = () => message.transcribing && !props.microphoneMuted + const pulse = () => + !props.animate || !transcribing() ? 0 : Math.max(props.userLevel(), transcriptionPulse(props.now()) * 0.12) + return ( + + + {transcribing() ? ( + + {props.animate ? meter(Math.max(0.12, pulse()), props.now() / 240) : "..."} + + ) : null} + + 0.74 + pulse() * 0.2} + /> + + + + ) + } + if (message.kind === "assistant") { + const pulse = () => (!props.animate || !message.streaming ? 0.5 : props.assistantLevel()) + return ( + + + (message.streaming ? 1 : 0.7)} + /> + + + ) + } + if (message.kind === "tool") + return ( + + + + {message.output === undefined ? ( + props.animate ? ( + + ) : ( + + ) + ) : ( + + )} + + + {message.name} + {" "} + {toolSummary(message.name, message.output)} + + + {props.details ? ( + + + {(token) => {token.text}} + + + ) : null} + + ) + return ( + + + · + + + {message.text} + + + ) +} + +export async function createVoiceTUI(options: { + onInterrupt(): void + onExit(): void + onCycleVoice(): void + onToggleMicrophone(): void + onToggleSpeaker(): void + reducedMotion?: boolean +}): Promise { + let view = initialVoiceView() + const [messageState, setMessageState] = createStore({ items: [...view.messages] }) + const [status, setStatus] = createSignal({}) + const [details, setDetails] = createSignal(false) + const [animationFrame, setAnimationFrame] = createSignal(performance.now()) + let userActive = false + let animationTimer: ReturnType | undefined + let assistantSegments: Array<{ start: number; end: number; level: number }> = [] + let assistantSegmentIndex = 0 + let assistantScheduledUntil = 0 + let assistantLevel = 0 + let userTargetLevel = 0 + let userLevel = 0 + let userLevelAt = 0 + const pendingTools = new Map< + string, + { + name: string + input: unknown + output?: unknown + completed: boolean + showAt: number + timer: ReturnType + } + >() + const pendingMetaTimers = new Set>() + + const renderer = await createCliRenderer({ + useMouse: true, + // Handle Ctrl-C in OpenTUI's native input parser, before Solid handlers + // and rendering work. onDestroy performs process/audio cleanup below. + exitOnCtrlC: true, + exitSignals: [], + autoFocus: false, + openConsoleOnError: false, + screenMode: "alternate-screen", + externalOutputMode: "passthrough", + consoleMode: "disabled", + onDestroy: () => { + if (animationTimer) clearInterval(animationTimer) + pendingTools.forEach((tool) => clearTimeout(tool.timer)) + pendingMetaTimers.forEach(clearTimeout) + options.onExit() + }, + }) + + const applyView = (event: VoiceViewEvent) => { + const next = transitionVoiceView(view, event) + if (next === view) return false + view = next + setMessageState("items", reconcile([...view.messages], { key: "key" })) + return true + } + + const startAnimation = () => { + if (options.reducedMotion) return + if (animationTimer) return + animationTimer = setInterval(() => { + const now = performance.now() + while (assistantSegments[assistantSegmentIndex]?.end <= now) assistantSegmentIndex += 1 + const segment = assistantSegments[assistantSegmentIndex] + const output = segment?.start <= now ? segment : undefined + const assistantTarget = output?.level ?? 0 + const userTarget = now - userLevelAt < AUDIO_LEVEL_STALE_MS ? userTargetLevel : 0 + assistantLevel += (assistantTarget - assistantLevel) * (assistantTarget > assistantLevel ? 0.5 : 0.16) + userLevel += (userTarget - userLevel) * (userTarget > userLevel ? 0.5 : 0.18) + setAnimationFrame(now) + if (view.revealAnimationEndsAt > 0 && now >= view.revealAnimationEndsAt) applyView({ type: "reveals.completed" }) + const transcribing = + !status().microphoneMuted && + messageState.items.some((message) => message.kind === "user" && message.transcribing) + if ( + userActive || + transcribing || + now < view.revealAnimationEndsAt || + assistantSegmentIndex < assistantSegments.length || + assistantLevel > 0.01 || + userLevel > 0.01 + ) + return + clearInterval(animationTimer) + animationTimer = undefined + }, PCM_METER_FRAME_MS) + } + + const currentAssistantLevel = () => { + animationFrame() + return assistantLevel + } + const currentUserLevel = () => { + animationFrame() + return userLevel + } + + function App() { + useKeyboard((evt) => { + if (evt.ctrl && evt.name === "c") return + if (evt.repeated || evt.ctrl || evt.meta || evt.option || evt.shift) return + if (evt.name === "v") { + evt.preventDefault() + return options.onCycleVoice() + } + if (evt.name === "m") { + evt.preventDefault() + return options.onToggleMicrophone() + } + if (evt.name === "s") { + evt.preventDefault() + return options.onToggleSpeaker() + } + if (evt.name === "d") { + evt.preventDefault() + setDetails((current) => !current) + return + } + if (evt.name === "escape") { + evt.preventDefault() + options.onInterrupt() + } + }) + const runtimeLine = () => + [ + status().audio ?? "connecting…", + status().microphoneMuted ? "mic muted" : undefined, + status().speakerMuted ? "speaker muted" : undefined, + status().voice, + status().model, + status().project, + status().session ?? "no session", + ] + .filter(Boolean) + .join(" ") + return ( + + + + + + {(message) => ( + + )} + + + + + + + {runtimeLine()} + + + esc interrupt {" "} + v voice {" "} + m mic {" "} + s speaker {" "} + d details {details() ? "on" : "off"} {" "} + ctrl+c quit + + + + ) + } + + await render(() => , renderer) + + const pushMeta = (text: string) => { + if (pendingTools.size === 0) return void applyView({ type: "meta", text }) + const showAt = Math.max(...[...pendingTools.values()].map((tool) => tool.showAt)) + const timer = setTimeout( + () => { + pendingMetaTimers.delete(timer) + applyView({ type: "meta", text }) + }, + Math.max(0, showAt - performance.now()) + 1, + ) + pendingMetaTimers.add(timer) + } + + return { + meta: pushMeta, + userSpeaking: (active) => { + if (active) applyView({ type: "user.started" }) + userActive = active + if (active) startAnimation() + if (!active) userTargetLevel = 0 + }, + userReset: () => { + userActive = false + userTargetLevel = 0 + userLevel = 0 + applyView({ type: "user.reset" }) + }, + userAudioLevel: (level) => { + if (level === undefined) { + userLevelAt = 0 + return + } + userTargetLevel = Math.max(0, Math.min(1, level)) + userLevelAt = performance.now() + if (userActive) startAnimation() + }, + userCommitted: (itemID) => { + userActive = false + applyView({ type: "user.committed", itemID }) + startAnimation() + }, + userTranscript: (itemID, text, final = true) => { + applyView({ + type: "user.transcript", + itemID, + text, + final, + now: performance.now(), + animate: !options.reducedMotion, + }) + if (!final || view.revealAnimationEndsAt > 0) startAnimation() + }, + assistantAudio: (level, durationMs) => { + if (options.reducedMotion) return + if (assistantSegmentIndex > 512) { + assistantSegments = assistantSegments.slice(assistantSegmentIndex) + assistantSegmentIndex = 0 + } + const now = performance.now() + const start = Math.max(now, assistantScheduledUntil) + const end = start + durationMs + assistantSegments.push({ start, end, level: Math.max(0, Math.min(1, level)) }) + assistantScheduledUntil = end + startAnimation() + }, + assistantPlaybackStopped: () => { + assistantSegments = [] + assistantSegmentIndex = 0 + assistantScheduledUntil = 0 + assistantLevel = 0 + setAnimationFrame(performance.now()) + }, + assistantDelta: (text) => { + applyView({ + type: "assistant.delta", + text, + now: performance.now(), + animate: !options.reducedMotion, + }) + if (view.revealAnimationEndsAt > 0) startAnimation() + }, + assistantTranscript: (text) => { + applyView({ + type: "assistant.transcript", + text, + now: performance.now(), + animate: !options.reducedMotion, + }) + if (view.revealAnimationEndsAt > 0) startAnimation() + }, + assistantDone: () => { + assistantSegments = [] + assistantSegmentIndex = 0 + assistantScheduledUntil = 0 + assistantLevel = 0 + applyView({ type: "assistant.done" }) + }, + toolStart: (callID, name, input) => { + if ( + pendingTools.has(callID) || + view.messages.some((message) => message.kind === "tool" && message.callID === callID) + ) + return + const pending = { + name, + input, + completed: false, + showAt: performance.now() + TOOL_DISPLAY_DELAY_MS, + timer: setTimeout(() => { + const current = pendingTools.get(callID) + if (!current) return + pendingTools.delete(callID) + applyView({ + type: "tool.started", + callID, + name: current.name, + input: current.input, + }) + if (current.completed) applyView({ type: "tool.done", callID, output: current.output }) + }, TOOL_DISPLAY_DELAY_MS), + } + pendingTools.set(callID, pending) + }, + toolDone: (callID, output) => { + const pending = pendingTools.get(callID) + if (pending) { + pending.output = output + pending.completed = true + return + } + applyView({ type: "tool.done", callID, output }) + }, + setStatus: (patch) => { + setStatus((current) => ({ ...current, ...patch })) + }, + close: () => { + if (!renderer.isDestroyed) renderer.destroy() + }, + } +} diff --git a/packages/voice/src/voice-coordinator.ts b/packages/voice/src/voice-coordinator.ts new file mode 100644 index 0000000000..ed6ae98779 --- /dev/null +++ b/packages/voice/src/voice-coordinator.ts @@ -0,0 +1,146 @@ +export type VoiceNotification = { + readonly promptID?: string + readonly text: string +} + +export type VoiceState = { + readonly connection: "connecting" | "ready" | "closed" + readonly conversation: "idle" | "waiting" | "responding" + readonly assistant: "idle" | "active" | "suppressed" + readonly userSpeaking: boolean + readonly tools: ReadonlySet + readonly notifications: ReadonlyArray + readonly desiredVoice: string + readonly reconnect: "none" | "pending" +} + +export type VoiceEvent = + | { readonly type: "connection.ready" } + | { readonly type: "connection.connecting" } + | { readonly type: "connection.closed" } + | { readonly type: "user.started" } + | { readonly type: "user.stopped" } + | { readonly type: "user.committed" } + | { readonly type: "assistant.started" } + | { readonly type: "assistant.suppressed" } + | { readonly type: "assistant.done"; readonly awaitingWork: boolean } + | { readonly type: "tool.started"; readonly id: string } + | { readonly type: "tool.finished"; readonly id: string } + | { readonly type: "notification.queued"; readonly notification: VoiceNotification } + | { readonly type: "notification.failed"; readonly notification: VoiceNotification } + | { readonly type: "voice.selected"; readonly voice: string } + +export type VoiceCommand = + | { readonly type: "notification.send"; readonly notification: VoiceNotification } + | { readonly type: "connection.reconnect" } + +export type VoiceTransition = { + readonly state: VoiceState + readonly commands: ReadonlyArray +} + +export function initialVoiceState(voice: string): VoiceState { + return { + connection: "connecting", + conversation: "idle", + assistant: "idle", + userSpeaking: false, + tools: new Set(), + notifications: [], + desiredVoice: voice, + reconnect: "none", + } +} + +export function transitionVoice(state: VoiceState, event: VoiceEvent): VoiceTransition { + const next = reduce(state, event) + return settle(next) +} + +function reduce(state: VoiceState, event: VoiceEvent): VoiceState { + switch (event.type) { + case "connection.ready": + if (state.connection === "ready") return state + return { ...state, connection: "ready" } + case "connection.connecting": + return { + ...state, + connection: "connecting", + conversation: "idle", + assistant: "idle", + userSpeaking: false, + } + case "connection.closed": + return { ...state, connection: "closed", conversation: "idle", assistant: "idle", userSpeaking: false } + case "user.started": + if (state.userSpeaking) return state + return { ...state, userSpeaking: true } + case "user.stopped": + if (!state.userSpeaking) return state + return { ...state, userSpeaking: false } + case "user.committed": + return { ...state, userSpeaking: false, conversation: "waiting", assistant: "idle" } + case "assistant.started": + if (state.assistant === "suppressed" || state.assistant === "active") return state + return { ...state, conversation: "responding", assistant: "active" } + case "assistant.suppressed": + return { ...state, assistant: "suppressed" } + case "assistant.done": + return { + ...state, + conversation: event.awaitingWork ? "waiting" : "idle", + assistant: "idle", + } + case "tool.started": + if (state.tools.has(event.id)) return state + return { ...state, conversation: "waiting", tools: new Set([...state.tools, event.id]) } + case "tool.finished": { + if (!state.tools.has(event.id)) return state + const tools = new Set(state.tools) + tools.delete(event.id) + return { ...state, tools } + } + case "notification.queued": + return { ...state, notifications: [...state.notifications, event.notification] } + case "notification.failed": + return { + ...state, + connection: "ready", + conversation: "idle", + notifications: [event.notification, ...state.notifications], + reconnect: "pending", + } + case "voice.selected": + if (event.voice === state.desiredVoice) return state + return { ...state, desiredVoice: event.voice, reconnect: "pending" } + } +} + +function settle(state: VoiceState): VoiceTransition { + if ( + state.reconnect === "pending" && + state.connection === "ready" && + state.conversation === "idle" && + state.tools.size === 0 + ) + return { + state: { ...state, connection: "connecting", reconnect: "none" }, + commands: [{ type: "connection.reconnect" }], + } + + if ( + state.notifications.length > 0 && + state.connection === "ready" && + state.conversation === "idle" && + !state.userSpeaking && + state.tools.size === 0 + ) { + const notification = state.notifications[0] + return { + state: { ...state, conversation: "waiting", notifications: state.notifications.slice(1) }, + commands: [{ type: "notification.send", notification }], + } + } + + return { state, commands: [] } +} diff --git a/packages/voice/test/animation.test.ts b/packages/voice/test/animation.test.ts new file mode 100644 index 0000000000..7f49137a4f --- /dev/null +++ b/packages/voice/test/animation.test.ts @@ -0,0 +1,29 @@ +import { describe, expect, test } from "bun:test" +import { scheduleTextReveal, springOpacity, transcriptionPulse } from "../src/animation" + +describe("voice text animation", () => { + test("stages newly streamed words without delaying them indefinitely", () => { + const first = scheduleTextReveal("", "Hello there", 1_000, 0) + expect(first.reveals).toEqual([ + { offset: 0, at: 1_000 }, + { offset: 6, at: 1_032 }, + ]) + expect(first.animationEndsAt).toBe(1_352) + + const queued = scheduleTextReveal("Hello there", " friend", 1_010, 10_000) + expect(queued.reveals).toEqual([{ offset: 12, at: 1_170 }]) + }) + + test("fades text monotonically to fully visible", () => { + expect(springOpacity(100, 100)).toBe(0) + expect(springOpacity(260, 100)).toBeGreaterThan(0) + expect(springOpacity(260, 100)).toBeLessThan(1) + expect(springOpacity(420, 100)).toBe(1) + }) + + test("keeps transcription pulse bounded and moving", () => { + const levels = [0, 110, 220, 330].map(transcriptionPulse) + expect(levels.every((level) => level >= 0 && level <= 1)).toBe(true) + expect(new Set(levels).size).toBeGreaterThan(1) + }) +}) diff --git a/packages/voice/test/assistant-text.test.ts b/packages/voice/test/assistant-text.test.ts new file mode 100644 index 0000000000..dacbd61620 --- /dev/null +++ b/packages/voice/test/assistant-text.test.ts @@ -0,0 +1,79 @@ +import { describe, expect, test } from "bun:test" +import { joinAssistantText } from "../src/ui-model" + +describe("joinAssistantText", () => { + const cases: Array<[name: string, left: string, right: string, expected: string]> = [ + [ + "inserts a space between sentences", + "Okay, I'll check that.", + "The build passed.", + "Okay, I'll check that. The build passed.", + ], + ["inserts a space mid-sentence", "Let me look at", "the config file.", "Let me look at the config file."], + [ + "keeps a single trailing space", + "Okay, I'll check that. ", + "The build passed.", + "Okay, I'll check that. The build passed.", + ], + [ + "keeps a single leading space", + "Okay, I'll check that.", + " The build passed.", + "Okay, I'll check that. The build passed.", + ], + ["collapses spaces on both sides", "Done. ", " Next.", "Done. Next."], + ["collapses runs of spaces", "Done. ", "\t \tNext.", "Done. Next."], + [ + "spaces after a question mark", + "Want me to run tests?", + "I can do that now.", + "Want me to run tests? I can do that now.", + ], + ["spaces after an ellipsis", "Thinking…", "done.", "Thinking… done."], + ["preserves a trailing newline", "Done:\n", "Next up, tests.", "Done:\nNext up, tests."], + ["preserves a leading newline", "Done:", "\nNext up, tests.", "Done:\nNext up, tests."], + [ + "preserves a paragraph break", + "First paragraph.\n\n", + "Second paragraph.", + "First paragraph.\n\nSecond paragraph.", + ], + ["preserves a newline mixed with spaces", "Done: ", " \n Next.", "Done: \n Next."], + ["never spaces before closing punctuation", "Done", ".", "Done."], + ["never spaces before a comma", "one", ", two", "one, two"], + ["never spaces before a closing paren", "(aside", ")", "(aside)"], + ["never spaces after an opening paren", "an (", "aside)", "an (aside)"], + ["drops an empty left side", "", "The build passed.", "The build passed."], + ["drops an empty right side", "The build passed.", "", "The build passed."], + ["drops a whitespace-only left side", " ", "The build passed.", "The build passed."], + ["drops a whitespace-only right side", "The build passed.", " \n ", "The build passed."], + ["handles two empty sides", "", "", ""], + ["does not trim the outer edges", " Leading kept.", "Trailing kept. ", " Leading kept. Trailing kept. "], + ] + + for (const [name, left, right, expected] of cases) { + test(name, () => expect(joinAssistantText(left, right).text).toBe(expected)) + } + + test("splits the result at the boundary so reveals cover only the appended text", () => { + for (const [name, left, right] of cases) { + const joined = joinAssistantText(left, right) + expect(`${name}: ${joined.previous}${joined.appended}`).toBe(`${name}: ${joined.text}`) + expect(`${name}: ${joined.text.endsWith(joined.appended)}`).toBe(`${name}: true`) + } + }) + + test("is idempotent once a boundary is normalised", () => { + const once = joinAssistantText("Okay.", "Next.").text + expect(joinAssistantText(once, "").text).toBe(once) + expect(joinAssistantText("Okay. ", "Next.").text).toBe(once) + }) + + test("folds consecutive messages with exactly one space per boundary", () => { + expect(["First.", "Second.", "Third."].reduce((left, right) => joinAssistantText(left, right).text)).toBe( + "First. Second. Third.", + ) + expect(["First.", "", "Third."].reduce((left, right) => joinAssistantText(left, right).text)).toBe("First. Third.") + }) +}) diff --git a/packages/voice/test/audio-jitter-buffer.test.ts b/packages/voice/test/audio-jitter-buffer.test.ts new file mode 100644 index 0000000000..4896ccb7c5 --- /dev/null +++ b/packages/voice/test/audio-jitter-buffer.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, test } from "bun:test" +import { AudioJitterBuffer } from "../src/audio-jitter-buffer" +import { PCM_BYTES_PER_MS } from "../src/pcm" + +const chunk = (durationMs: number, gapMs = 0) => ({ bytes: Buffer.alloc(durationMs * PCM_BYTES_PER_MS), gapMs }) + +describe("AudioJitterBuffer", () => { + test("holds startup audio until the target buffer is available", () => { + const buffer = new AudioJitterBuffer(300) + expect(buffer.push(chunk(100))).toEqual([]) + expect(buffer.push(chunk(100))).toEqual([]) + expect(buffer.push(chunk(100))).toHaveLength(3) + }) + + test("passes chunks through after playback starts", () => { + const buffer = new AudioJitterBuffer(200) + buffer.push(chunk(200)) + expect(buffer.push(chunk(100))).toEqual([chunk(100)]) + }) + + test("flushes short utterances and resets between responses", () => { + const buffer = new AudioJitterBuffer(300) + buffer.push(chunk(100)) + expect(buffer.finish()).toHaveLength(1) + buffer.reset() + expect(buffer.push(chunk(100))).toEqual([]) + }) + + test("counts intentional timeline silence toward buffered duration", () => { + const buffer = new AudioJitterBuffer(300) + expect(buffer.push(chunk(200, 100))).toHaveLength(1) + }) +}) diff --git a/packages/voice/test/completion-store.test.ts b/packages/voice/test/completion-store.test.ts new file mode 100644 index 0000000000..6012422504 --- /dev/null +++ b/packages/voice/test/completion-store.test.ts @@ -0,0 +1,50 @@ +import { expect, test } from "bun:test" +import { mkdtemp, rm } from "node:fs/promises" +import { tmpdir } from "node:os" +import { join } from "node:path" +import { createCompletionStore } from "../src/completion-store" + +test("completion store restores pending and completed notifications until delivery", async () => { + const directory = await mkdtemp(join(tmpdir(), "voice-completions-")) + const path = join(directory, "prompts.json") + const first = await createCompletionStore(path) + const pending = { sessionID: "session-1", promptID: "prompt-1" } + const completed = { sessionID: "session-2", promptID: "prompt-2" } + + try { + await first.admitting(pending, "Please continue") + expect(first.entries()).toEqual([{ status: "admitting", handle: pending, text: "Please continue" }]) + await first.pending(pending) + await first.pending(completed) + await first.completed(completed, { + type: "opencode.prompt.completed", + session_id: "session-2", + prompt_id: "prompt-2", + status: "completed", + text: "done", + }) + await first.close() + + const restored = await createCompletionStore(path) + expect(restored.entries()).toEqual([ + { status: "pending", handle: pending }, + { + status: "completed", + handle: completed, + notification: { + type: "opencode.prompt.completed", + session_id: "session-2", + prompt_id: "prompt-2", + status: "completed", + text: "done", + }, + }, + ]) + + await restored.delivered("prompt-2") + await restored.close() + expect((await createCompletionStore(path)).entries()).toEqual([{ status: "pending", handle: pending }]) + } finally { + await rm(directory, { recursive: true, force: true }) + } +}) diff --git a/packages/voice/test/live-projector.test.ts b/packages/voice/test/live-projector.test.ts new file mode 100644 index 0000000000..4b30421a0f --- /dev/null +++ b/packages/voice/test/live-projector.test.ts @@ -0,0 +1,268 @@ +import { describe, expect, test } from "bun:test" +import { createLiveEventProjector } from "../src/protocol-live" +import type { VoiceProtocolEvent } from "../src/protocol" + +const frames = [ + { type: "session.started", session: {} }, + { type: "session.context.appended", start_ms: 0, end_ms: 0 }, + { type: "output_audio.delta", audio: "AAA=", start_ms: 0, end_ms: 1 }, + { + type: "input_transcript.added", + start_ms: 0, + end_ms: 100, + item: { id: "input-1", type: "input_transcript", text: " Hello" }, + }, + { type: "turn.created", turn: { id: "user-turn", role: "user", transcript: " Hello" } }, + { + type: "input_transcript.added", + start_ms: 100, + end_ms: 200, + item: { id: "input-2", type: "input_transcript", text: ", this" }, + }, + { type: "turn.delta", turn_id: "user-turn", delta: ", this" }, + { + type: "output_transcript.added", + start_ms: 200, + end_ms: 300, + item: { id: "output-1", type: "output_transcript", text: " Copy that" }, + }, + { + type: "input_transcript.added", + start_ms: 200, + end_ms: 300, + item: { id: "input-3", type: "input_transcript", text: " is a test" }, + }, + { + type: "output_transcript.added", + start_ms: 300, + end_ms: 400, + item: { id: "output-2", type: "output_transcript", text: ", test" }, + }, + { type: "turn.delta", turn_id: "user-turn", delta: " is a test" }, + { type: "turn.done", turn: { id: "user-turn", role: "user", transcript: " Hello, this is a test" } }, + { type: "turn.created", turn: { id: "assistant-turn", role: "assistant", transcript: " Copy that, test" } }, + { + type: "output_transcript.added", + start_ms: 400, + end_ms: 500, + item: { id: "output-3", type: "output_transcript", text: " received" }, + }, + { type: "turn.delta", turn_id: "assistant-turn", delta: " received" }, + { + type: "output_transcript.added", + start_ms: 500, + end_ms: 600, + item: { id: "output-4", type: "output_transcript", text: " loud and clear." }, + }, + { type: "turn.delta", turn_id: "assistant-turn", delta: " loud and clear." }, + { + type: "turn.done", + turn: { id: "assistant-turn", role: "assistant", transcript: " Copy that, test received loud and clear." }, + }, +] as const + +describe("Live event projector", () => { + test("replays progressive transcripts without schema errors or duplicate text", () => { + const project = createLiveEventProjector() + const events = frames.flatMap((frame) => project(JSON.stringify(frame)).events) + + expect(events.filter((event) => event.type === "error")).toEqual([]) + expect(events.filter((event) => event.type === "user.committed")).toEqual([ + { type: "user.committed", id: "input-1" }, + ]) + expect(events.filter(isUserTranscript)).toEqual([ + { type: "user.transcript", id: "input-1", text: "Hello", final: false }, + { type: "user.transcript", id: "input-1", text: "Hello, this", final: false }, + { type: "user.transcript", id: "input-1", text: "Hello, this is a test", final: false }, + { type: "user.transcript", id: "input-1", text: "Hello, this is a test", final: true }, + ]) + expect( + events + .filter(isAssistantDelta) + .map((event) => event.delta) + .join(""), + ).toBe(" Copy that, test received loud and clear.") + expect(events.filter((event) => event.type === "assistant.done")).toEqual([ + { type: "assistant.done", awaitingWork: false }, + ]) + expect(events.filter((event) => event.type === "assistant.audio")).toHaveLength(1) + }) + + test("reports malformed frames instead of throwing", () => { + expect(createLiveEventProjector()(JSON.stringify({ type: "turn.created", turn: { id: 1 } })).events).toEqual([ + { type: "error", message: "Received an invalid Live API event." }, + ]) + }) + + test("replaces corrected assistant transcript snapshots", () => { + const project = createLiveEventProjector() + const events = [ + project( + JSON.stringify({ + type: "output_transcript.added", + item: { id: "output-1", type: "output_transcript", text: " Hello world" }, + }), + ), + project( + JSON.stringify({ + type: "turn.created", + turn: { id: "assistant-turn", role: "assistant", transcript: " Hello there" }, + }), + ), + ].flatMap((result) => result.events) + + expect(events.filter((event) => event.type.startsWith("assistant.transcript"))).toEqual([ + { type: "assistant.transcript.delta", delta: " Hello world" }, + { type: "assistant.transcript", text: " Hello there" }, + ]) + }) + + test("keeps the space between consecutive assistant turns", () => { + const project = createLiveEventProjector() + const turn = (id: string, text: string) => [ + { type: "turn.created", turn: { id, role: "assistant", transcript: "" } }, + { type: "output_transcript.added", item: { id: `${id}-o`, type: "output_transcript", text } }, + { type: "turn.delta", turn_id: id, delta: text }, + { type: "turn.done", turn: { id, role: "assistant", transcript: text } }, + ] + const events = [...turn("t1", ' Session "Fix auth bug" is ready.'), ...turn("t2", " Next up, tests.")].flatMap( + (frame) => project(JSON.stringify(frame)).events, + ) + + // Concatenating the delta stream is what the UI and any transcript consumer does; the + // turn boundary must survive it rather than welding "ready.Next" together. + const transcript = events + .filter(isAssistantDelta) + .map((event) => event.delta) + .join("") + expect(transcript).toBe(' Session "Fix auth bug" is ready. Next up, tests.') + expect(transcript).not.toContain("ready.Next") + expect(events.filter((event) => event.type === "assistant.done")).toHaveLength(2) + }) + + test("emits no duplicate text when fragments and turn deltas overlap across turns", () => { + const project = createLiveEventProjector() + const frames = [ + { type: "turn.created", turn: { id: "t1", role: "assistant", transcript: "" } }, + { type: "output_transcript.added", item: { id: "a", type: "output_transcript", text: " First" } }, + { type: "turn.delta", turn_id: "t1", delta: " First" }, + { type: "output_transcript.added", item: { id: "b", type: "output_transcript", text: " turn." } }, + { type: "turn.delta", turn_id: "t1", delta: " turn." }, + { type: "turn.done", turn: { id: "t1", role: "assistant", transcript: " First turn." } }, + { type: "turn.created", turn: { id: "t2", role: "assistant", transcript: "" } }, + { type: "output_transcript.added", item: { id: "c", type: "output_transcript", text: " Second turn." } }, + { type: "turn.delta", turn_id: "t2", delta: " Second turn." }, + { type: "turn.done", turn: { id: "t2", role: "assistant", transcript: " Second turn." } }, + ] + const events = frames.flatMap((frame) => project(JSON.stringify(frame)).events) + + expect(events.filter((event) => event.type === "assistant.transcript")).toEqual([]) + expect( + events + .filter(isAssistantDelta) + .map((event) => event.delta) + .join(""), + ).toBe(" First turn. Second turn.") + }) + + test("projects Responses delegation function calls without rejecting reasoning items", () => { + const project = createLiveEventProjector() + expect( + project( + JSON.stringify({ + type: "response.output_item.done", + item: { id: "reasoning-1", type: "reasoning" }, + }), + ).events, + ).toEqual([{ type: "debug", message: "response.output_item.done" }]) + expect( + project( + JSON.stringify({ + type: "response.output_item.added", + item: { + id: "function-1", + type: "function_call", + name: "find_sessions", + call_id: "call-1", + arguments: "", + }, + }), + ).events, + ).toEqual([ + { type: "debug", message: "response.output_item.added" }, + { type: "tool.started", id: "call-1", name: "find_sessions" }, + ]) + expect( + project( + JSON.stringify({ + type: "response.output_item.done", + item: { + id: "function-1", + type: "function_call", + name: "find_sessions", + call_id: "call-1", + arguments: '{"scope":"current_project"}', + }, + }), + ).events, + ).toEqual([ + { type: "debug", message: "response.output_item.done" }, + { + type: "work.requested", + request: { id: "call-1", name: "find_sessions", input: { scope: "current_project" } }, + }, + ]) + expect( + project( + JSON.stringify({ + type: "response.output_item.done", + item: { + id: "message-1", + type: "message", + status: "completed", + content: [{ type: "output_text", text: "Two sessions found." }], + role: "assistant", + }, + }), + ).events, + ).toEqual([{ type: "debug", message: "response.output_item.done" }]) + }) + + test("rejects malformed function arguments without hanging the delegation", () => { + expect( + createLiveEventProjector()( + JSON.stringify({ + type: "response.output_item.done", + item: { + id: "function-1", + type: "function_call", + name: "find_sessions", + call_id: "call-invalid", + arguments: "not json", + }, + }), + ).events, + ).toEqual([ + { type: "debug", message: "response.output_item.done" }, + { type: "tool.started", id: "call-invalid", name: "find_sessions" }, + { type: "error", message: "Received invalid arguments for Live tool find_sessions." }, + { + type: "work.rejected", + request: { id: "call-invalid", name: "find_sessions", input: {} }, + output: { status: "error", message: "Invalid arguments for tool find_sessions." }, + }, + ]) + }) +}) + +function isUserTranscript( + event: VoiceProtocolEvent, +): event is Extract { + return event.type === "user.transcript" +} + +function isAssistantDelta( + event: VoiceProtocolEvent, +): event is Extract { + return event.type === "assistant.transcript.delta" +} diff --git a/packages/voice/test/opencode.test.ts b/packages/voice/test/opencode.test.ts new file mode 100644 index 0000000000..afd5d48dcf --- /dev/null +++ b/packages/voice/test/opencode.test.ts @@ -0,0 +1,152 @@ +import { expect, test } from "bun:test" +import { OpenCode } from "@opencode-ai/client/promise" +import { createOpenCodeBridge } from "../src/opencode" +import type { CompletionStore } from "../src/completion-store" + +test("renames and archives only sessions discovered by the voice controller", async () => { + const requests: Request[] = [] + const server: { archived?: number } = {} + const session = { + id: "ses_known", + projectID: "project-1", + cost: "0", + tokens: { input: 0, output: 0, reasoning: 0, cache: { read: 0, write: 0 } }, + time: { created: 1, updated: 2 }, + title: "Old title", + location: { directory: "/workspace" }, + } + const fetch = Object.assign( + async (input: string | URL | Request, init?: RequestInit) => { + const request = input instanceof Request ? input : new Request(input.toString(), init) + requests.push(request) + const url = new URL(request.url) + if (url.pathname === "/api/event") + return new Response("", { headers: { "content-type": "text/event-stream" } }) + if (url.pathname === "/api/project") + return Response.json([ + { + id: "project-1", + worktree: "/workspace", + time: { created: 1, updated: 2 }, + sandboxes: [], + }, + ]) + if (url.pathname === "/api/session" && request.method === "GET") + return Response.json({ data: [session], cursor: {} }) + if (url.pathname === "/api/session/ses_known" && request.method === "GET") + return Response.json({ data: { ...session, time: { ...session.time, archived: server.archived } } }) + if (url.pathname === "/api/session/ses_known/rename" && request.method === "POST") + return new Response(null, { status: 204 }) + if (url.pathname === "/api/session/ses_known/archive" && request.method === "POST") { + server.archived = Date.now() + return new Response(null, { status: 204 }) + } + return new Response("Not found", { status: 404 }) + }, + { preconnect: () => {} }, + ) + const client = OpenCode.make({ + baseUrl: "https://opencode.test", + fetch, + }) + const bridge = await createOpenCodeBridge({ + client, + directory: "/workspace", + model: { providerID: "openai", id: "gpt-test" }, + notify: () => {}, + onSession: () => {}, + completionStore: memoryCompletionStore(), + }) + + try { + expect(bridge.definitions.find((tool) => tool.name === "rename_session")).toMatchObject({ + description: expect.stringContaining("does not prompt, wake, or interrupt"), + parameters: { + type: "object", + additionalProperties: false, + required: ["session_id", "title"], + }, + }) + expect(bridge.definitions.find((tool) => tool.name === "archive_session")).toMatchObject({ + description: expect.stringContaining("explicitly confirms"), + parameters: { + type: "object", + additionalProperties: false, + required: ["session_id"], + }, + }) + + expect(await bridge.execute("rename_session", { session_id: "ses_unknown", title: "Nope" })).toEqual({ + status: "error", + message: "Use a session ID returned by find_sessions or start_session.", + retryable: false, + }) + expect(await bridge.execute("archive_session", {})).toEqual({ + status: "error", + message: "Use a session ID returned by find_sessions or start_session.", + retryable: false, + }) + expect(await bridge.execute("archive_session", { session_id: "ses_unknown" })).toEqual({ + status: "error", + message: "Use a session ID returned by find_sessions or start_session.", + retryable: false, + }) + expect(requests.some((request) => new URL(request.url).pathname.includes("ses_unknown"))).toBe(false) + + await bridge.execute("find_sessions", { + query: null, + scope: "current_project", + recency: "any", + limit: 10, + }) + expect(await bridge.execute("rename_session", { session_id: "ses_known", title: " " })).toEqual({ + status: "error", + message: "A non-empty session title is required.", + retryable: false, + }) + expect(await bridge.execute("rename_session", { session_id: "ses_known", title: "Old title" })).toEqual({ + status: "unchanged", + session_id: "ses_known", + previous_title: "Old title", + title: "Old title", + }) + expect(requests.some((request) => new URL(request.url).pathname.endsWith("/rename"))).toBe(false) + + expect(await bridge.execute("rename_session", { session_id: "ses_known", title: " New title " })).toEqual({ + status: "renamed", + session_id: "ses_known", + previous_title: "Old title", + title: "New title", + }) + const rename = requests.find((request) => new URL(request.url).pathname.endsWith("/rename")) + expect(rename?.method).toBe("POST") + expect(await rename?.json()).toEqual({ title: "New title" }) + + expect(await bridge.execute("archive_session", { session_id: "ses_known" })).toEqual({ + status: "archived", + title: "Old title", + }) + const archive = requests.find((request) => new URL(request.url).pathname.endsWith("/archive")) + expect(archive?.method).toBe("POST") + expect(await archive?.text()).toBe("") + expect(await bridge.execute("archive_session", { session_id: "ses_known" })).toEqual({ + status: "already_archived", + title: "Old title", + }) + expect(requests.filter((request) => new URL(request.url).pathname.endsWith("/archive"))).toHaveLength(1) + expect(requests.some((request) => /\/(prompt|interrupt)$/.test(new URL(request.url).pathname))).toBe(false) + } finally { + await bridge.close() + } +}) + +function memoryCompletionStore(): CompletionStore { + return { + entries: () => [], + admitting: async () => {}, + pending: async () => {}, + completed: async () => {}, + delivered: async () => {}, + close: async () => {}, + } +} diff --git a/packages/voice/test/pcm.test.ts b/packages/voice/test/pcm.test.ts new file mode 100644 index 0000000000..e3144d47bd --- /dev/null +++ b/packages/voice/test/pcm.test.ts @@ -0,0 +1,10 @@ +import { expect, test } from "bun:test" +import { pcmLevel } from "../src/pcm" + +test("PCM level ignores silence and normalizes audible energy", () => { + expect(pcmLevel(Buffer.alloc(4_800))).toBe(0) + const audible = Buffer.alloc(4_800) + for (let offset = 0; offset < audible.length; offset += 2) audible.writeInt16LE(8_000, offset) + expect(pcmLevel(audible)).toBeGreaterThan(0.8) + expect(pcmLevel(audible)).toBeLessThanOrEqual(1) +}) diff --git a/packages/voice/test/ui-layout.test.ts b/packages/voice/test/ui-layout.test.ts new file mode 100644 index 0000000000..887d7e3c21 --- /dev/null +++ b/packages/voice/test/ui-layout.test.ts @@ -0,0 +1,241 @@ +import { expect, mock, test } from "bun:test" +import { createTestRenderer } from "@opentui/core/testing" + +test("voice rows fill the viewport, wrap under content, and keep tools after their preamble", async () => { + const setup = await createTestRenderer({ width: 64, height: 18, useThread: false }) + const core = await import("@opentui/core") + void mock.module("@opentui/core", () => ({ ...core, createCliRenderer: async () => setup.renderer })) + const { createVoiceTUI } = await import("../src/ui") + let interrupts = 0 + let voiceCycles = 0 + let microphoneToggles = 0 + let speakerToggles = 0 + const ui = await createVoiceTUI({ + onInterrupt: () => interrupts++, + onExit: () => {}, + onCycleVoice: () => voiceCycles++, + onToggleMicrophone: () => microphoneToggles++, + onToggleSpeaker: () => speakerToggles++, + reducedMotion: true, + }) + + try { + ui.setStatus({ audio: "duplex", voice: "marin", model: "gpt-live" }) + ui.meta("[session] created session-1") + ui.userCommitted("user-1") + ui.userTranscript("user-1", "Please inspect the current session", true) + ui.toolStart("tool-1", "opencode", { request: "inspect" }) + ui.meta("[session] adopted session-1") + ui.assistantDelta("Let me check. This sentence is") + ui.assistantDelta(" long enough to wrap cleanly under its column.") + ui.assistantDone() + ui.assistantDelta("Continued in the same assistant row.") + ui.assistantDone() + await Bun.sleep(200) + await setup.flush() + + const rows = frameRows(setup.captureCharFrame()) + expect(rows).toHaveLength(18) + expect(rows.every((row) => row.length === 64)).toBe(true) + expect(rows[16]).toStartWith(" duplex marin gpt-live") + expect(rows[17]).toContain("esc interrupt") + expect(rows[17]).not.toContain("any key") + expect(rows.filter((row) => row.includes("Let me check"))).toHaveLength(1) + expect(rows.findIndex((row) => row.includes("opencode"))).toBeGreaterThan( + rows.findIndex((row) => row.includes("Let me check")), + ) + expect(rows.findIndex((row) => row.includes("[session] adopted"))).toBeGreaterThan( + rows.findIndex((row) => row.includes("opencode")), + ) + expect(rows.find((row) => row.includes("under its column"))?.indexOf("under")).toBe( + rows.find((row) => row.includes("Let me check"))?.indexOf("Let"), + ) + // Deltas inside one turn concatenate raw; a delta after assistantDone starts a new + // message and must gain exactly one space at the boundary. + const assistant = rows.join("") + expect(assistant).toContain("This sentence is long enough") + expect(assistant).toContain("column. Continued") + expect(assistant).not.toContain("column.Continued") + expect(rows[0]).toStartWith(" · [session]") + + setup.resize(96, 22) + await setup.flush() + const wide = frameRows(setup.captureCharFrame()) + expect(wide).toHaveLength(22) + expect(wide.every((row) => row.length === 96)).toBe(true) + expect(wide.at(-2)).toStartWith(" duplex marin gpt-live") + + setup.resize(48, 16) + await setup.flush() + const narrow = frameRows(setup.captureCharFrame()) + expect(narrow).toHaveLength(16) + expect(narrow.every((row) => row.length === 48)).toBe(true) + expect(narrow.at(-1)).toContain("esc interrupt") + + setup.resize(64, 18) + await setup.flush() + + ui.userSpeaking(true) + await setup.flush() + const active = setup + .captureCharFrame() + .split("\n") + .find((row) => row.includes("│ ...")) + expect(active).toContain("│ ...") + expect(active).not.toContain("listening") + expect(active).not.toContain("you") + + ui.userReset() + await setup.flush() + expect(setup.captureCharFrame()).not.toContain("│ ...") + + ui.userSpeaking(true) + ui.userSpeaking(false) + ui.userCommitted("user-2") + ui.userTranscript("user-2", "partial transcript", false) + await setup.flush() + const partial = setup + .captureCharFrame() + .split("\n") + .find((row) => row.includes("partial transcript")) + expect(partial).toContain("│ ... partial transcript") + expect(partial).not.toContain("you") + + ui.userSpeaking(true) + await setup.flush() + expect( + setup + .captureCharFrame() + .split("\n") + .filter((row) => row.includes("│ ...")), + ).toHaveLength(2) + + ui.userReset() + await setup.flush() + + ui.setStatus({ microphoneMuted: true }) + await setup.flush() + const muted = setup + .captureCharFrame() + .split("\n") + .find((row) => row.includes("partial transcript")) + expect(muted).toContain("│ partial transcript") + expect(muted).not.toContain("│ ...") + + ui.userSpeaking(false) + ui.setStatus({ microphoneMuted: false }) + ui.userTranscript("user-2", "partial transcript", true) + await setup.flush() + const complete = setup + .captureCharFrame() + .split("\n") + .find((row) => row.includes("partial transcript")) + expect(complete).toContain("│ partial transcript") + + ui.assistantDelta("First response.") + ui.userCommitted("boundary-user") + ui.userTranscript("boundary-user", "A new request", true) + ui.assistantDelta("Second response.") + await setup.flush() + const boundaries = setup.captureCharFrame().split("\n") + expect(boundaries.findIndex((row) => row.includes("First response."))).toBeLessThan( + boundaries.findIndex((row) => row.includes("A new request")), + ) + expect(boundaries.findIndex((row) => row.includes("A new request"))).toBeLessThan( + boundaries.findIndex((row) => row.includes("Second response.")), + ) + + await setup.mockInput.typeText(" x") + setup.mockInput.pressArrow("right") + await setup.flush() + expect(interrupts).toBe(0) + expect(voiceCycles).toBe(0) + expect(microphoneToggles).toBe(0) + expect(speakerToggles).toBe(0) + + setup.mockInput.pressKey("v") + setup.mockInput.pressKey("m") + setup.mockInput.pressKey("s") + await setup.flush() + expect(voiceCycles).toBe(1) + expect(microphoneToggles).toBe(1) + expect(speakerToggles).toBe(1) + + setup.mockInput.pressEscape() + await Bun.sleep(50) + await setup.flush() + expect(interrupts).toBe(1) + } finally { + ui.close() + mock.restore() + } +}) + +// In audio mode ui.assistantDone() is scheduled off the playback clock (spike.ts finishPlayback), +// so the next turn's transcript deltas arrive while the row is still streaming. The turn boundary +// therefore has to survive in the delta text itself, not in the row's streaming flag. +test("keeps the turn boundary when the next turn streams before playback finishes", async () => { + const setup = await createTestRenderer({ width: 72, height: 12, useThread: false }) + const core = await import("@opentui/core") + void mock.module("@opentui/core", () => ({ ...core, createCliRenderer: async () => setup.renderer })) + const { createVoiceTUI } = await import("../src/ui") + const ui = await createVoiceTUI({ + onInterrupt: () => {}, + onExit: () => {}, + onCycleVoice: () => {}, + onToggleMicrophone: () => {}, + onToggleSpeaker: () => {}, + reducedMotion: true, + }) + + try { + // Verbatim projector deltas: each turn's first fragment carries the boundary space. + ui.assistantDelta(' Session "Fix auth bug"') + ui.assistantDelta(" is ready.") + // turn.done reaches the protocol, but playback is still draining, so no assistantDone() yet. + ui.assistantDelta(" Next up, tests.") + await setup.flush() + + const frame = setup.captureCharFrame() + expect(frame).toContain('Session "Fix auth bug" is ready. Next up, tests.') + expect(frame).not.toContain("ready.Next") + // The row itself must not open with the boundary space. + expect(frame).toContain('│ Session "Fix auth bug"') + expect(frame).not.toContain('│ Session "Fix auth bug"') + + ui.assistantDone() + await setup.flush() + expect(setup.captureCharFrame()).toContain("is ready. Next up, tests.") + } finally { + ui.close() + mock.restore() + } +}) + +test("keeps animated text mounted after its reveal completes", async () => { + const setup = await createTestRenderer({ width: 72, height: 10, useThread: false }) + const core = await import("@opentui/core") + void mock.module("@opentui/core", () => ({ ...core, createCliRenderer: async () => setup.renderer })) + const { createVoiceTUI } = await import("../src/ui") + const ui = await createVoiceTUI({ + onInterrupt: () => {}, + onExit: () => {}, + onCycleVoice: () => {}, + onToggleMicrophone: () => {}, + onToggleSpeaker: () => {}, + }) + + try { + ui.assistantDelta("Animated text remains visible.") + await Bun.sleep(500) + await setup.flush() + expect(setup.captureCharFrame()).toContain("Animated text remains visible.") + } finally { + ui.close() + mock.restore() + } +}) + +function frameRows(frame: string) { + return frame.split("\n").slice(0, -1) +} diff --git a/packages/voice/test/ui-model.test.ts b/packages/voice/test/ui-model.test.ts new file mode 100644 index 0000000000..65c11ea185 --- /dev/null +++ b/packages/voice/test/ui-model.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, test } from "bun:test" +import { initialVoiceView, transitionVoiceView, type VoiceViewEvent, type VoiceViewState } from "../src/ui-model" + +const apply = (state: VoiceViewState, ...events: ReadonlyArray) => + events.reduce(transitionVoiceView, state) + +describe("VoiceView", () => { + test("keeps assistant responses separated by a committed user turn", () => { + const state = apply( + initialVoiceView(), + { type: "assistant.delta", text: "First response.", now: 0, animate: false }, + { type: "user.committed", itemID: "user-1" }, + { type: "user.transcript", itemID: "user-1", text: "Next request", final: true, now: 1, animate: false }, + { type: "assistant.delta", text: "Second response.", now: 2, animate: false }, + ) + expect(state.messages.map((message) => message.kind)).toEqual(["assistant", "user", "assistant"]) + expect(state.messages.flatMap((message) => ("text" in message ? [message.text] : []))).toEqual([ + "First response.", + "Next request", + "Second response.", + ]) + }) + + test("adopts one provisional user row and removes an empty abandoned row", () => { + const started = transitionVoiceView(initialVoiceView(), { type: "user.started" }) + expect(started.messages).toHaveLength(1) + const committed = transitionVoiceView(started, { type: "user.committed", itemID: "user-1" }) + expect(committed.messages).toEqual([ + { key: "message-1", kind: "user", itemID: "user-1", transcribing: true, reveals: [] }, + ]) + + const abandoned = apply(initialVoiceView(), { type: "user.started" }, { type: "user.reset" }) + expect(abandoned.messages).toEqual([]) + }) + + test("deduplicates transcript snapshots and clears completed reveals", () => { + const state = apply( + initialVoiceView(), + { type: "user.committed", itemID: "user-1" }, + { type: "user.transcript", itemID: "user-1", text: "Hello", final: false, now: 0, animate: true }, + { type: "user.transcript", itemID: "user-1", text: "Hello", final: false, now: 10, animate: true }, + ) + expect(state.messages).toHaveLength(1) + expect(state.messages[0]?.kind === "user" ? state.messages[0].reveals.length : 0).toBeGreaterThan(0) + expect(transitionVoiceView(state, { type: "reveals.completed" }).messages).toEqual([ + { key: "message-1", kind: "user", itemID: "user-1", text: "Hello", transcribing: true, reveals: [] }, + ]) + }) + + test("bounds message history", () => { + const state = Array.from({ length: 205 }, (_, index) => index).reduce( + (current, index) => transitionVoiceView(current, { type: "meta", text: String(index) }), + initialVoiceView(), + ) + expect(state.messages).toHaveLength(200) + expect(state.messages[0]).toEqual({ key: "message-6", kind: "meta", text: "5" }) + }) + + test("settles tool completion before or after insertion", () => { + const inserted = apply( + initialVoiceView(), + { type: "tool.started", callID: "call-1", name: "read_session", input: {} }, + { type: "tool.done", callID: "call-1", output: { status: "ok" } }, + ) + expect(inserted.messages[0]).toEqual({ + key: "message-1", + kind: "tool", + callID: "call-1", + name: "read_session", + input: {}, + output: { status: "ok" }, + }) + }) +}) diff --git a/packages/voice/test/voice-coordinator.test.ts b/packages/voice/test/voice-coordinator.test.ts new file mode 100644 index 0000000000..c34104bba1 --- /dev/null +++ b/packages/voice/test/voice-coordinator.test.ts @@ -0,0 +1,102 @@ +import { describe, expect, test } from "bun:test" +import { initialVoiceState, transitionVoice, type VoiceEvent, type VoiceState } from "../src/voice-coordinator" + +const apply = (state: VoiceState, ...events: ReadonlyArray) => + events.reduce( + (result, event) => { + const next = transitionVoice(result.state, event) + return { state: next.state, commands: [...result.commands, ...next.commands] } + }, + { state, commands: [] as Array["commands"][number]> }, + ) + +describe("VoiceCoordinator", () => { + test("delivers one notification at a time when the server becomes idle", () => { + const notification = { promptID: "prompt-1", text: "completed" } + const waiting = apply( + initialVoiceState("marin"), + { type: "connection.ready" }, + { type: "user.committed" }, + { type: "notification.queued", notification }, + ) + expect(waiting.commands).toEqual([]) + + expect(transitionVoice(waiting.state, { type: "assistant.done", awaitingWork: false })).toEqual({ + state: { ...waiting.state, conversation: "waiting", assistant: "idle", notifications: [] }, + commands: [{ type: "notification.send", notification }], + }) + }) + + test("keeps notifications queued while the user is speaking", () => { + const notification = { text: "completed" } + const result = apply( + initialVoiceState("marin"), + { type: "connection.ready" }, + { type: "user.started" }, + { type: "notification.queued", notification }, + ) + expect(result.commands).toEqual([]) + expect(result.state.notifications).toEqual([notification]) + expect(transitionVoice(result.state, { type: "user.stopped" }).commands).toEqual([ + { type: "notification.send", notification }, + ]) + }) + + test("keeps notifications queued until active tool calls finish", () => { + const notification = { text: "completed" } + const result = apply( + initialVoiceState("marin"), + { type: "connection.ready" }, + { type: "tool.started", id: "call-1" }, + { type: "assistant.done", awaitingWork: false }, + { type: "notification.queued", notification }, + ) + expect(result.commands).toEqual([]) + expect(transitionVoice(result.state, { type: "tool.finished", id: "call-1" }).commands).toEqual([ + { type: "notification.send", notification }, + ]) + }) + + test("tracks tools by call ID and reconnects only after work and conversation settle", () => { + const busy = apply( + initialVoiceState("marin"), + { type: "connection.ready" }, + { type: "tool.started", id: "call-1" }, + { type: "voice.selected", voice: "cedar" }, + ) + expect(busy.commands).toEqual([]) + expect(transitionVoice(busy.state, { type: "tool.finished", id: "call-1" }).commands).toEqual([]) + + const idle = transitionVoice(transitionVoice(busy.state, { type: "tool.finished", id: "call-1" }).state, { + type: "assistant.done", + awaitingWork: false, + }) + expect(idle.commands).toEqual([{ type: "connection.reconnect" }]) + expect(idle.state.connection).toBe("connecting") + }) + + test("a user commit closes assistant suppression and establishes a waiting boundary", () => { + const suppressed = apply( + initialVoiceState("marin"), + { type: "connection.ready" }, + { type: "assistant.started" }, + { type: "assistant.suppressed" }, + { type: "user.committed" }, + ) + expect(suppressed.state.assistant).toBe("idle") + expect(suppressed.state.conversation).toBe("waiting") + }) + + test("requeues a notification when the adapter rejects it", () => { + const notification = { text: "completed" } + const sent = apply( + initialVoiceState("marin"), + { type: "connection.ready" }, + { type: "notification.queued", notification }, + ) + const failed = transitionVoice(sent.state, { type: "notification.failed", notification }) + expect(failed.state.notifications).toEqual([notification]) + expect(failed.state.connection).toBe("connecting") + expect(failed.commands).toEqual([{ type: "connection.reconnect" }]) + }) +}) diff --git a/packages/voice/tsconfig.json b/packages/voice/tsconfig.json new file mode 100644 index 0000000000..82d77bf1e5 --- /dev/null +++ b/packages/voice/tsconfig.json @@ -0,0 +1,11 @@ +{ + "$schema": "https://json.schemastore.org/tsconfig", + "extends": "@tsconfig/bun/tsconfig.json", + "compilerOptions": { + "jsx": "preserve", + "jsxImportSource": "@opentui/solid", + "noUncheckedIndexedAccess": false, + "noUnusedLocals": true + }, + "include": ["src", "test"] +}