From 0a2adfa9f6d472941b7494a03c8a34a7a0901b18 Mon Sep 17 00:00:00 2001 From: Eliot Lim Date: Sun, 4 Oct 2026 04:52:02 +0800 Subject: [PATCH 1/4] feat(server): add managed local whisper transcription (MEET-3) --- _brief-meet3.md | 29 +++++ packages/sdk/src/ai.ts | 2 + packages/server/src/ai/service.ts | 11 +- packages/server/src/ai/whisper.test.ts | 70 ++++++++++++ packages/server/src/ai/whisper.ts | 130 ++++++++++++++++++++++ packages/server/src/server.ts | 4 +- packages/server/src/transcription.test.ts | 54 +++++++++ packages/ui/src/components/AiSettings.tsx | 11 +- 8 files changed, 306 insertions(+), 5 deletions(-) create mode 100644 _brief-meet3.md create mode 100644 packages/server/src/ai/whisper.test.ts create mode 100644 packages/server/src/ai/whisper.ts diff --git a/_brief-meet3.md b/_brief-meet3.md new file mode 100644 index 00000000..f9a5881b --- /dev/null +++ b/_brief-meet3.md @@ -0,0 +1,29 @@ +# MEET-3 — Local whisper engine (default transcription path) + +You are Finley, an OpenBook Worker agent. Work in THIS worktree (`/Users/eliot/Workspaces/OpenBook-wt-meet-2`). Do NOT push. Conventional commits (`feat(server): … (MEET-3)`), committed incrementally. Disk headroom is ~6 GiB — prefer a SMALL whisper model for tests/default; clean temp artifacts. + +## Branch setup +`git checkout -b feat/meet-3-local-whisper` from current HEAD (feat/meet-2-transcribe @ bbccdbc1). This stacks on MEET-2; merge order to main is MEET-2 first. + +## Task +Implement the LOCAL transcription backend so transcription works with zero cloud keys — this is the product DEFAULT (owner directive). MEET-2 built the exact seam for you; its contract: + +- Implement `TranscriptionEngine` from `packages/server/src/ai/providers.ts`: `transcribe(bytes: Uint8Array, opts?: {filename?, mime?, signal?}): Promise` ({text, segments?[{start,end,text}] seconds, durationMs}). Respect AbortSignal. +- Inject via the optional third `AiService` constructor argument `() => Promise` (resolve/start the managed backend lazily; return null when unavailable). MEET-3 owns process/model lifecycle; wire the resolver at AiService construction in server startup. +- Resolution order already implemented by MEET-2: explicit cloud > your local resolver > mock > actionable 400. Local is ungated (no paid gate). Usage rows log kind 'transcribe' provider 'local', cost 0. + +## Implementation guidance +- Follow the `LlamaEngine` optional-native-dep pattern (providers.ts ~:468, node-llama-cpp): whisper binding as an OPTIONAL dependency — evaluate smart-whisper vs whisper.cpp node addons vs spawning a whisper.cpp/whisper-server binary; pick fewest native-build headaches across mac/linux CI and justify in your report. An OpenAI-compatible local server (reusing `new OpenAiCompatEngine(baseUrl, model)` against a managed localhost process) is also acceptable if it's the most robust path. +- Model acquisition reuses the existing model-download flow (ai/service.ts startDownload pattern, OPENBOOK_MODELS_DIR, server.ts ~:437). Default model: whisper base or small multilingual (ggml); state the size/quality trade-off. Do NOT bundle model bytes in git. +- Absence of the native dep / model must degrade to a clear error pointing at Settings → AI model download — never a crash. CI (`pnpm verify`) must be green WITHOUT the native dep installed (mock/skip pattern like llama). +- Settings → AI: expose local transcription state (model present/absent, download affordance) through the existing aiStatus/config surfaces; keep UI changes minimal — MEET-5's Settings polish is out of scope. + +## Acceptance (each maps to a test) +1. With the binding available + a model present and no cloud config, POST /api/ai/transcribe returns real text+segments for a small fixture (check in a fixture only if <100 KB; else synthesize audio in the test and assert non-error shape). Gate this test behind an env flag or dep-presence check so CI without the dep skips it EXPLICITLY (visible skip, not silent). +2. Resolver returns null when dep/model missing → route falls through per MEET-2 order; actionable 400 text mentions Settings → AI. +3. AbortSignal cancels an in-flight local transcription. +4. Usage row: provider 'local', cost 0. +5. `pnpm verify` green FOREGROUND (VITEST_MAX_WORKERS=1 if the money.test.ts load flake appears — do not modify that test), WITHOUT the native dep in the default run. + +## Done +All committed, not pushed. Write `_report-meet3.md`: outcome first, head sha, dep choice + rationale, model default + trade-off, criterion→test map, deviations, open questions. Delete nothing from existing tests. Never poll external state; wedged after a real attempt → commit + report. diff --git a/packages/sdk/src/ai.ts b/packages/sdk/src/ai.ts index 5212ad79..1a0662e6 100644 --- a/packages/sdk/src/ai.ts +++ b/packages/sdk/src/ai.ts @@ -212,6 +212,8 @@ export interface AiUsageResponse { } export interface AiStatus { + /** Local audio capability is independent of the chat provider. */ + transcription?: {model: string; modelPresent: boolean; runtimeAvailable: boolean; ready: boolean; downloadUrl: string; detail?: string}; config: AiConfig; /** The engine can generate text right now. */ ready: boolean; diff --git a/packages/server/src/ai/service.ts b/packages/server/src/ai/service.ts index f03187b1..de55c935 100644 --- a/packages/server/src/ai/service.ts +++ b/packages/server/src/ai/service.ts @@ -5,6 +5,7 @@ import {providerSettings, type AiConfig, type AiProvider, type AiProviderSetting import type {Db} from '../db'; import {createEngine, MockEngine, OpenAiCompatEngine, type TranscriptionEngine, type AiEngine, type GenerateOptions} from './providers'; import {assembleSearchResults, bm25Scores, buildIndex, cosine, pageRowsToDocs, parseTaskList, type Bm25Index} from './search'; +import {WHISPER_MODEL, WHISPER_MODEL_URL} from './whisper'; import {SkillStore} from './skills'; /** @@ -114,6 +115,7 @@ export class AiService { private readonly modelsDir: string, /** MEET-3: lazily resolve the managed local audio backend. */ private readonly localTranscription?: () => Promise, + private readonly localLifecycle?: {status(): Promise>; dispose(): Promise}, ) { this.skills = new SkillStore(db); } @@ -162,9 +164,9 @@ export class AiService { return {engine: new OpenAiCompatEngine(audio.baseUrl?.trim() || 'https://api.openai.com', audio.model?.trim() || 'whisper-1', 'openai', audio.apiKey), provider: 'openai-compat', model: audio.model?.trim() || 'whisper-1'}; } const local = await this.localTranscription?.(); - if (local) return {engine: local, provider: 'local', model: audio?.model ?? 'local'}; + if (local) return {engine: local, provider: 'local', model: WHISPER_MODEL}; if (config.provider === 'mock') return {engine: new MockEngine(), provider: 'mock', model: 'mock'}; - throw new TranscriptionConfigError('Local transcription is unavailable. Configure transcription in Settings → AI.'); + throw new TranscriptionConfigError('Local transcription is unavailable. Download the Whisper model and check the local runtime in Settings → AI.'); } async status(): Promise { @@ -183,6 +185,7 @@ export class AiService { } return { config: this.config, + transcription: await this.localLifecycle?.status(), ready, embeddings, detail, @@ -337,6 +340,7 @@ export class AiService { await this.loadConfig(); if (this.download && !this.download.done && !this.download.error) return this.download; const fileName = decodeURIComponent(new URL(url).pathname.split('/').pop() || 'model.gguf'); + if (fileName !== path.basename(fileName) || fileName === '.' || fileName === '..' || fileName.includes('\\')) throw new Error('Invalid model filename'); mkdirSync(this.modelsDir, {recursive: true}); const dest = path.join(this.modelsDir, fileName); const state: DownloadState = {url, received: 0, total: null, done: false}; @@ -367,7 +371,7 @@ export class AiService { state.done = true; } // Auto-select the downloaded model for the llama provider. - if (this.config.provider === 'llama' && !this.config.model) { + if (url !== WHISPER_MODEL_URL && this.config.provider === 'llama' && !this.config.model) { await this.setConfig({...this.config, model: fileName}); } } catch (err) { @@ -381,5 +385,6 @@ export class AiService { async dispose(): Promise { await this.engine?.dispose().catch(() => undefined); + await this.localLifecycle?.dispose(); } } diff --git a/packages/server/src/ai/whisper.test.ts b/packages/server/src/ai/whisper.test.ts new file mode 100644 index 00000000..d310b787 --- /dev/null +++ b/packages/server/src/ai/whisper.test.ts @@ -0,0 +1,70 @@ +import {mkdtemp, readFile, readdir, rm, writeFile} from 'node:fs/promises'; +import {tmpdir} from 'node:os'; +import path from 'node:path'; +import {afterEach, beforeEach, describe, expect, it} from 'vitest'; +import {LocalWhisper, parseWhisperOutput, runWhisperProcess, WHISPER_MODEL} from './whisper'; + +let dir: string; +beforeEach(async () => { dir = await mkdtemp(path.join(tmpdir(), 'meet3-test-')); }); +afterEach(async () => { await rm(dir, {recursive: true, force: true}); }); + +async function script(name: string, source: string): Promise { + const file = path.join(dir, name); + await writeFile(file, `#!${process.execPath}\n${source}`, {mode: 0o700}); + return file; +} + +describe('optional local whisper runtime', () => { + it('returns null for missing model or executables, and discovers a model without restart', async () => { + const local = new LocalWhisper(dir, process.execPath, process.execPath); + expect(await local.resolve()).toBeNull(); + expect(await local.status()).toMatchObject({modelPresent: false, runtimeAvailable: true, ready: false}); + await writeFile(path.join(dir, WHISPER_MODEL), 'test model'); + expect(await local.resolve()).toBe(local); + const missing = new LocalWhisper(dir, path.join(dir, 'missing'), process.execPath); + expect(await missing.resolve()).toBeNull(); + expect(await missing.status()).toMatchObject({modelPresent: true, runtimeAvailable: false, detail: expect.stringContaining('Settings → AI')}); + await local.dispose(); + expect(await local.resolve()).toBeNull(); + }); + + it('decodes exact input bytes, converts millisecond offsets, and cleans scratch files', async () => { + const marker = path.join(dir, 'scratch'); + const ffmpeg = await script('ffmpeg', 'const fs = require(\'node:fs\'); const args = process.argv.slice(2); const input = args[args.indexOf(\'-i\') + 1]; if (fs.readFileSync(input).toString() !== \'recording\') process.exit(2); fs.writeFileSync(args.at(-1), \'wav\');'); + const whisper = await script('whisper', `const fs = require('node:fs'); const args = process.argv.slice(2); const output = args[args.indexOf('-of') + 1]; fs.writeFileSync(${JSON.stringify(marker)}, require('node:path').dirname(output)); fs.writeFileSync(output + '.json', JSON.stringify({transcription: [{offsets: {from: 250, to: 1500}, text: ' Hello world.'}]}));`); + const local = new LocalWhisper(dir, whisper, ffmpeg); + expect(await local.transcribe(new TextEncoder().encode('recording'), {filename: '../../escape.webm', mime: 'audio/webm'})) + .toEqual({text: 'Hello world.', segments: [{start: 0.25, end: 1.5, text: 'Hello world.'}], durationMs: 1500}); + await expect(readdir(await readFile(marker, 'utf8'))).rejects.toThrow(); + }); + + it.each(['abort', 'dispose'] as const)('%s kills in-flight inference and removes scratch files', async (action) => { + const marker = path.join(dir, 'started'); + const ffmpeg = await script('ffmpeg', 'process.exit(0);'); + const whisper = await script('whisper', `const fs = require('node:fs'); const args = process.argv.slice(2); fs.writeFileSync(${JSON.stringify(marker)}, JSON.stringify({pid: process.pid, dir: require('node:path').dirname(args[args.indexOf('-of') + 1])})); setInterval(() => {}, 1000);`); + const local = new LocalWhisper(dir, whisper, ffmpeg); + const controller = new AbortController(); + const promise = local.transcribe(new Uint8Array([1]), {signal: controller.signal}); + const rejected = expect(promise).rejects.toMatchObject({name: 'AbortError'}); + await expect.poll(async () => readFile(marker, 'utf8').catch(() => '')).not.toBe(''); + const started = JSON.parse(await readFile(marker, 'utf8')) as {pid: number; dir: string}; + if (action === 'abort') controller.abort(); + else await local.dispose(); + await rejected; + expect(() => process.kill(started.pid, 0)).toThrow(); + await expect(readdir(started.dir)).rejects.toThrow(); + }); + + it('rejects pre-aborted work, spawn failures, failed processes and malformed output', async () => { + const controller = new AbortController(); + controller.abort(); + await expect(new LocalWhisper(dir).transcribe(new Uint8Array(), {signal: controller.signal})).rejects.toMatchObject({name: 'AbortError'}); + const signal = new AbortController().signal; + await expect(runWhisperProcess(path.join(dir, 'missing'), [], signal)).rejects.toThrow(); + await expect(runWhisperProcess(process.execPath, ['-e', 'process.exit(1)'], signal)).rejects.toThrow('Local audio processing failed'); + for (const value of [null, {}, {transcription: [{text: 'bad', offsets: {from: -1, to: 100}}]}]) { + expect(() => parseWhisperOutput(value)).toThrow(); + } + expect(parseWhisperOutput({transcription: []})).toEqual({text: '', segments: [], durationMs: 0}); + }); +}); diff --git a/packages/server/src/ai/whisper.ts b/packages/server/src/ai/whisper.ts new file mode 100644 index 00000000..bec7d535 --- /dev/null +++ b/packages/server/src/ai/whisper.ts @@ -0,0 +1,130 @@ +import {spawn} from 'node:child_process'; +import {constants} from 'node:fs'; +import {access, mkdtemp, readFile, rm, stat, writeFile} from 'node:fs/promises'; +import {tmpdir} from 'node:os'; +import path from 'node:path'; +import type {AiStatus, AiTranscriptionResult} from '@book.dev/sdk'; +import type {TranscriptionEngine, TranscribeOptions} from './providers'; + +export const WHISPER_MODEL = 'ggml-base.bin'; +export const WHISPER_MODEL_URL = 'https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.bin'; + +async function executable(command: string): Promise { + const candidates = path.isAbsolute(command) || command.includes(path.sep) + ? [command] : (process.env.PATH ?? '').split(path.delimiter).map((dir) => path.join(dir, command)); + for (const candidate of candidates) { + try { + await access(candidate, constants.X_OK); + if ((await stat(candidate)).isFile()) return candidate; + } catch { /* Optional system dependency. */ } + } + return null; +} + +/** One child per operation: no listening port or native Node ABI dependency. + * SIGKILL makes cancellation deterministic even inside a native inference call. + * Wait for close before removing its private scratch directory. */ +export async function runWhisperProcess(command: string, args: string[], signal: AbortSignal): Promise { + signal.throwIfAborted(); + await new Promise((resolve, reject) => { + const child = spawn(command, args, {stdio: 'ignore', shell: false}); + const abort = () => { child.kill('SIGKILL'); }; + signal.addEventListener('abort', abort, {once: true}); + if (signal.aborted) abort(); + child.once('error', (error) => { + signal.removeEventListener('abort', abort); + reject(error); + }); + child.once('close', (code) => { + signal.removeEventListener('abort', abort); + if (signal.aborted) reject(signal.reason); + else if (code !== 0) reject(new Error('Local audio processing failed. Check the recording and Settings → AI.')); + else resolve(); + }); + }); +} + +export function parseWhisperOutput(value: unknown): AiTranscriptionResult { + const data = value as {transcription?: {offsets?: {from?: number; to?: number}; text?: string}[]}; + if (!data || !Array.isArray(data.transcription)) throw new Error('Invalid whisper output'); + const segments = data.transcription.map((segment) => { + const start = segment?.offsets?.from; + const end = segment?.offsets?.to; + if (typeof start !== 'number' || !Number.isFinite(start) || start < 0 + || typeof end !== 'number' || !Number.isFinite(end) || end < start || typeof segment.text !== 'string') { + throw new Error('Invalid whisper segment'); + } + return {start: start / 1000, end: end / 1000, text: segment.text.trim()}; + }); + return {text: segments.map((s) => s.text).join(' ').trim(), segments, durationMs: Math.max(0, ...segments.map((s) => s.end * 1000))}; +} + +/** Optional whisper.cpp + FFmpeg runtime. Models live beside chat models, but + * inference runs only on demand, releasing model memory after each recording. */ +export class LocalWhisper implements TranscriptionEngine { + private readonly active = new Set(); + private readonly pending = new Set>(); + private disposed = false; + + constructor( + private readonly modelsDir: string, + private readonly whisperCommand = process.env.OPENBOOK_WHISPER_BIN || 'whisper-cli', + private readonly ffmpegCommand = process.env.OPENBOOK_FFMPEG_BIN || 'ffmpeg', + ) {} + + async status(): Promise> { + const [whisper, ffmpeg, modelPresent] = await Promise.all([ + executable(this.whisperCommand), executable(this.ffmpegCommand), + stat(path.join(this.modelsDir, WHISPER_MODEL)).then((s) => s.isFile() && s.size > 0).catch(() => false), + ]); + const runtimeAvailable = Boolean(whisper && ffmpeg); + return { + model: WHISPER_MODEL, modelPresent, runtimeAvailable, ready: modelPresent && runtimeAvailable && !this.disposed, + downloadUrl: WHISPER_MODEL_URL, + detail: !runtimeAvailable ? 'Install whisper.cpp (whisper-cli) and FFmpeg on the server, then return to Settings → AI.' + : !modelPresent ? 'Download Whisper base in Settings → AI to enable local transcription.' : undefined, + }; + } + + async resolve(): Promise { + return (await this.status()).ready ? this : null; + } + + transcribe(bytes: Uint8Array, opts: TranscribeOptions = {}): Promise { + const controller = new AbortController(); + const signal = opts.signal ? AbortSignal.any([opts.signal, controller.signal]) : controller.signal; + this.active.add(controller); + const operation = this.perform(bytes, signal).finally(() => { + this.active.delete(controller); + this.pending.delete(operation); + }); + this.pending.add(operation); + return operation; + } + + private async perform(bytes: Uint8Array, signal: AbortSignal): Promise { + signal.throwIfAborted(); + if (this.disposed) throw new Error('Local transcription has stopped.'); + const dir = await mkdtemp(path.join(tmpdir(), 'openbook-whisper-')); + try { + const input = path.join(dir, 'input'); + const wav = path.join(dir, 'audio.wav'); + const output = path.join(dir, 'result'); + await writeFile(input, bytes, {signal}); + // Probe by content, never by an untrusted filename; accept browser WebM, + // MP4, Ogg and WAV alike. Restrict nested input protocols to local files. + await runWhisperProcess(this.ffmpegCommand, ['-nostdin', '-v', 'error', '-protocol_whitelist', 'file,pipe', '-i', input, '-vn', '-ar', '16000', '-ac', '1', '-c:a', 'pcm_s16le', wav], signal); + await runWhisperProcess(this.whisperCommand, ['-m', path.join(this.modelsDir, WHISPER_MODEL), '-f', wav, '-l', 'auto', '-oj', '-of', output], signal); + signal.throwIfAborted(); + return parseWhisperOutput(JSON.parse(await readFile(`${output}.json`, 'utf8'))); + } finally { + await rm(dir, {recursive: true, force: true}); + } + } + + async dispose(): Promise { + this.disposed = true; + for (const controller of this.active) controller.abort(); + await Promise.allSettled(this.pending); + } +} diff --git a/packages/server/src/server.ts b/packages/server/src/server.ts index bfbff2ba..a236505e 100644 --- a/packages/server/src/server.ts +++ b/packages/server/src/server.ts @@ -6,6 +6,7 @@ import {PageStore, PAGE_VERSION_KEEP, PAGE_VERSION_MAX_AGE_MS} from './store'; import {PageHub} from './hub'; import {BookMirror, MirrorLockedError, WriteBudgetError} from './mirror'; import {AiService} from './ai/service'; +import {LocalWhisper} from './ai/whisper'; import {McpClientManager} from './ai/mcpClients'; import {AiUsageLog} from './ai/usage'; import {IdentityService} from './instanceConfig'; @@ -436,7 +437,8 @@ export async function startServer(opts: StartOptions): Promise { // (server mode). The subsystem is inert until configured via /api/ai. const modelsDir = process.env.OPENBOOK_MODELS_DIR || (opts.dataDir ? path.join(opts.dataDir, 'models') : path.join(os.homedir(), '.openbook', 'models')); - const ai = new AiService(db, modelsDir); + const whisper = new LocalWhisper(modelsDir); + const ai = new AiService(db, modelsDir, () => whisper.resolve(), whisper); // External-tools (MCP client) manager (AGENT-3): owned beside AiService, pools // connections to admin-registered MCP servers and hands the agent route // namespaced `mcp__*` tools. Inert until an admin configures + enables a server diff --git a/packages/server/src/transcription.test.ts b/packages/server/src/transcription.test.ts index 6d6fce60..b0216a34 100644 --- a/packages/server/src/transcription.test.ts +++ b/packages/server/src/transcription.test.ts @@ -11,6 +11,8 @@ import {AiService} from './ai/service'; import {MockEngine, OpenAiCompatEngine} from './ai/providers'; import {AiUsageLog} from './ai/usage'; import {LocalDataClient} from './localClient'; +import {LocalWhisper, WHISPER_MODEL, WHISPER_MODEL_URL} from './ai/whisper'; +import {readFile, readdir} from 'node:fs/promises'; let db: PgliteDb; let store: PageStore; @@ -119,6 +121,58 @@ describe('transcription contract', () => { expect(local).toHaveBeenCalledOnce(); }); + it('uses the managed resolver fallback and exposes actionable local state through aiStatus', async () => { + const local = new LocalWhisper(dir, join(dir, 'missing'), join(dir, 'missing-ffmpeg')); + const service = new AiService(db, dir, () => local.resolve(), local); + const app = appWith(service); + expect((await post(app)).status).toBe(400); + expect((await (await post(app)).json()).error).toContain('Settings → AI'); + const status = await app.request(API.aiStatus, {headers: {...headers, [LOCAL_OWNER_HEADER]: secret}}); + expect((await status.json()).transcription).toMatchObject({modelPresent: false, runtimeAvailable: false, ready: false, downloadUrl: WHISPER_MODEL_URL}); + await service.setConfig({provider: 'mock'}); + expect((await post(app)).status).toBe(200); + await service.dispose(); + }); + + it('logs local transcription as free and downloads its model without selecting it for chat', async () => { + const service = new AiService(db, dir, async () => new MockEngine()); + const usage = new AiUsageLog(store); + expect((await post(appWith(service, usage))).status).toBe(200); + expect((await usage.report()).rows?.[0]).toMatchObject({provider: 'local', model: WHISPER_MODEL, kind: 'transcribe', cost: 0}); + await service.setConfig({provider: 'llama'}); + vi.stubGlobal('fetch', vi.fn(async () => new Response('model bytes'))); + const app = appWith(service); + const download = await app.request(API.aiModelDownload, {method: 'POST', headers: {...headers, [LOCAL_OWNER_HEADER]: secret}, body: JSON.stringify({url: WHISPER_MODEL_URL})}); + expect(download.status).toBe(200); + await expect.poll(async () => (await service.status()).download?.done).toBe(true); + expect(await readFile(join(dir, WHISPER_MODEL), 'utf8')).toBe('model bytes'); + expect((await service.getConfig()).model).toBeUndefined(); + expect(await readdir(dir)).not.toContain(`${WHISPER_MODEL}.part`); + await service.dispose(); + }); + + // Opt-in: OPENBOOK_TEST_WHISPER=1, OPENBOOK_MODELS_DIR containing ggml-base.bin, + // whisper-cli + ffmpeg on PATH (or OPENBOOK_WHISPER_BIN / OPENBOOK_FFMPEG_BIN). + it.skipIf(process.env.OPENBOOK_TEST_WHISPER !== '1')('native whisper: POST transcribes synthesized WAV with no cloud keys', async () => { + const local = new LocalWhisper(process.env.OPENBOOK_MODELS_DIR || join(dir, 'models')); + expect((await local.status()).ready).toBe(true); + const wav = Buffer.alloc(44 + 16000 * 2); + wav.write('RIFF'); wav.writeUInt32LE(wav.length - 8, 4); wav.write('WAVEfmt ', 8); + wav.writeUInt32LE(16, 16); wav.writeUInt16LE(1, 20); wav.writeUInt16LE(1, 22); + wav.writeUInt32LE(16000, 24); wav.writeUInt32LE(32000, 28); wav.writeUInt16LE(2, 32); wav.writeUInt16LE(16, 34); + wav.write('data', 36); wav.writeUInt32LE(wav.length - 44, 40); + assetId = (await store.putAsset(wav, 'audio/wav')).id; + await store.refAsset(assetId, pageId); + const service = new AiService(db, dir, () => local.resolve(), local); + try { + const response = await post(appWith(service)); + expect(response.status).toBe(200); + expect(await response.json()).toMatchObject({text: expect.any(String), segments: expect.any(Array), durationMs: expect.any(Number)}); + } finally { + await service.dispose(); + } + }, 120_000); + it('returns identical 404s for missing, unreadable, unreferenced, and unrelated assets/pages', async () => { await ai.setConfig({provider: 'mock'}); await store.updateInstanceConfig({ownerSubject: 'test#owner', guestAccess: 'read'}); diff --git a/packages/ui/src/components/AiSettings.tsx b/packages/ui/src/components/AiSettings.tsx index d4b868c1..da02bb8a 100644 --- a/packages/ui/src/components/AiSettings.tsx +++ b/packages/ui/src/components/AiSettings.tsx @@ -27,7 +27,7 @@ function normalize(c: AiConfig): AiConfig { // the `apiKeySet` signal so the form knows a key is stored without holding it. providers[c.provider] = {model: c.model, baseUrl: c.baseUrl, apiKeySet: c.apiKeySet, autoStart: c.autoStart}; } - return {provider: c.provider, providers, effort: c.effort, thinking: c.thinking}; + return {provider: c.provider, providers, effort: c.effort, thinking: c.thinking, transcription: c.transcription}; } /** @@ -273,6 +273,15 @@ export default function AiSettings() { return ( + {status?.transcription && ( + +

{status.transcription.modelPresent ? 'Model downloaded.' : 'Model not downloaded.'} {status.transcription.ready ? 'Ready to transcribe.' : status.transcription.detail}

+ + {download?.url === status.transcription.downloadUrl && download.error &&

{download.error}

} +
+ )}
{providers.map((p) => ( From 67c08f09c0cc5cf7b776aacb1f62756258a828f9 Mon Sep 17 00:00:00 2001 From: Eliot Lim Date: Sun, 4 Oct 2026 06:57:36 +0800 Subject: [PATCH 2/4] docs(server): report local whisper verification (MEET-3) --- _report-meet3.md | 62 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 62 insertions(+) create mode 100644 _report-meet3.md diff --git a/_report-meet3.md b/_report-meet3.md new file mode 100644 index 00000000..149f1db7 --- /dev/null +++ b/_report-meet3.md @@ -0,0 +1,62 @@ +# MEET-3 — Finley report + +Implemented the default local transcription path, managed process cleanup, model download/status, and the minimal Settings → AI affordance. No cloud key is required. Full foreground `pnpm verify` passed, as did native HTTP and real-speech smoke checks. Nothing was pushed. + +- Branch: `feat/meet-3-local-whisper`, stacked on `feat/meet-2-transcribe` at `bbccdbc1`. +- Implementation HEAD: `0a2adfa9f6d472941b7494a03c8a34a7a0901b18`. The subsequent report-only commit records verification; its SHA is supplied in the worker response. +- Merge MEET-2 before MEET-3. + +## Runtime choice and model + +Chose the optional **whisper.cpp `whisper-cli` system executable**, plus **FFmpeg** for browser WebM/MP4/Ogg and other audio inputs. No new npm native addon or mandatory install/build hook is introduced. Compared with [smart-whisper's native Node addon](https://github.com/JacobLinCool/smart-whisper), the subprocess approach avoids Node ABI coupling and addon build failures in macOS/Linux CI. Compared with a persistent whisper-server, it needs no port allocation, readiness polling, or resident model memory. The cost is loading the model for each recording and requiring host-installed binaries. + +The [upstream whisper.cpp CLI](https://github.com/ggml-org/whisper.cpp/tree/master/examples/cli) supports JSON output. Its millisecond offsets become API segments in seconds; their final endpoint supplies `durationMs`, following MEET-2's segment-derived timing fallback. FFmpeg normalizes input to mono 16 kHz PCM WAV. The server spawns commands directly without a shell, uses private temporary directories, ignores caller filenames, kills work on abort/shutdown, and removes scratch files after children close. + +Default: **multilingual Whisper base**, `ggml-base.bin`, approximately **142 MiB**. It is a smaller download and uses less memory than small/medium models, with lower accuracy on noisy speech, accents, and difficult multilingual recordings. Language detection is automatic. Model bytes are never committed. + +Setup: + +1. Install whisper.cpp (`whisper-cli`) and FFmpeg on the server host using the [upstream build instructions](https://github.com/ggml-org/whisper.cpp#quick-start) or the host package manager. +2. Make both executables available on the server's PATH. Alternatively set `OPENBOOK_WHISPER_BIN` and `OPENBOOK_FFMPEG_BIN` to executable paths before starting OpenBook. +3. In Settings → AI, select **Download Whisper base**. The existing authenticated model-download flow downloads into `OPENBOOK_MODELS_DIR`, or the server's existing default models directory. The local resolver discovers the completed model without restart. +4. Local is the transcription default even when chat is off. Explicit cloud configuration still takes precedence, then local, then the existing mock fallback. Explicit transcription off stays off. + +## Criterion → test map + +| Acceptance | Evidence | +| --- | --- | +| 1. Native local POST, no cloud keys, result with segments | `transcription.test.ts`: `native whisper: POST transcribes synthesized WAV with no cloud keys`. Explicitly gated by `OPENBOOK_TEST_WHISPER=1`; synthesizes a 32 KB WAV in memory and asserts non-error result shape. **Passed in a separate opt-in native run** with temporary binaries/model; default verification still skips it explicitly. An additional real-speech sample returned nonempty text and timed segments. | +| 2. Missing runtime/model → null, fallback, actionable 400 | `ai/whisper.test.ts`: missing model/executable and discovery-without-restart test. `transcription.test.ts`: managed resolver fallback and aiStatus test; existing off/unavailable/cloud-precedence tests retained. | +| 3. Abort cancels in-flight local inference | `ai/whisper.test.ts`: actual child-process cancellation and disposal tests; both verify the child is gone and scratch directory removed. Also tests pre-aborted work and process failures. | +| 4. Usage provider local, cost 0 | `transcription.test.ts`: local usage/download test asserts `provider: local`, `model: ggml-base.bin`, `kind: transcribe`, `cost: 0`. | +| 5. Foreground verify, no native dependency | Verification results below. | + +Additional tests cover exact input-byte handling, timestamp normalization, malformed JSON shape, empty transcription, temporary-file cleanup, owner-route model download, and preventing Whisper downloads from becoming the llama chat model. No existing tests were deleted. + +Native smoke command after installing the optional binaries/model: + +```sh +OPENBOOK_TEST_WHISPER=1 OPENBOOK_MODELS_DIR=/absolute/path/to/models \ + pnpm --filter @book.dev/server exec vitest run src/transcription.test.ts +``` + +## Verification + +- Focused tests: **17 passed, 1 explicitly skipped** across the transcription and whisper suites. +- Implementation commit hooks: ESLint, SDK/server/UI typechecks, and conventional commit validation passed. +- Opt-in native HTTP smoke: **1 passed** (12 unrelated tests skipped by the name filter), 11.11 seconds. Temporary CPU-only whisper.cpp build at `60c0be6ac8fa71b1a2ae2dd938a31a34a508e774`, FFmpeg reporting version 6.0, and multilingual base on macOS arm64. +- Real-speech smoke: the upstream temporary `samples/jfk.wav` produced the expected “ask not” text, a segment spanning 0–10.5 seconds, and `durationMs: 10500`. No sample/model bytes were added to git. +- Native build tools, binaries, source, model, and scratch files were removed after the smoke runs. The isolated downloads/build occupied roughly 400 MiB and never entered the server's default PATH or model directory. +- First foreground `pnpm verify` inside the macOS sandbox: all earlier stages passed; server tests finished with 98 files / 1,342 tests passing, 7 skipped, and one mirror integration failure accompanied by 14 `EMFILE` filesystem-watch errors. The file-descriptor soft limit was already 1,048,575; a standalone `fs.watch` on a new empty temporary directory also emitted `EMFILE` inside the sandbox, confirming the environment restriction. +- The unchanged mirror integration suite passed all 3 tests outside the sandbox (14.43 seconds). No existing tests or resource limits were changed. +- Full foreground `pnpm verify` rerun outside the sandbox: **PASSED, exit 0**, with no optional Whisper runtime/model installed in the default environment. Builds, generated-file checks, typechecks, lint, all package tests, and end-to-end checks passed. SDK: 546 tests; UI: 2,319; app: 7; server: 99 files, 1,343 tests passed and 7 explicitly skipped. Server end-to-end: 256 checks; MCP end-to-end: 70 checks. The server unit suite took 3,397.77 seconds. + +Local verification artifacts: [verify-green.log](/Users/eliot/.bb/thread-storage/meet3-verification-gl8tvey3/verify-green.log), [verify-sandbox-failure.log](/Users/eliot/.bb/thread-storage/meet3-verification-gl8tvey3/verify-sandbox-failure.log), [native-http.log](/Users/eliot/.bb/thread-storage/meet3-verification-gl8tvey3/native-http.log), [native-speech.log](/Users/eliot/.bb/thread-storage/meet3-verification-gl8tvey3/native-speech.log). + +## Deviations and open questions + +- Used the brief's permitted optional executable alternative, not an npm `optionalDependencies` entry. CI requires neither whisper.cpp nor FFmpeg. Product packaging/distribution of those binaries remains a follow-up; Settings explicitly explains missing runtime requirements. +- The manager supplies status/disposal through a fourth optional `AiService` constructor argument, preserving the existing third-argument resolver contract and existing callers. +- No native runtime/model was installed into the workspace or globally. The default suite explicitly skips native inference; the separate opt-in run passed using temporary executables/model. Linux and GPU backends were not exercised here. +- Model downloads reuse the existing single-download slot and progress surface. The local model is fixed to base for this milestone; selecting larger local models and broader Settings polish remain outside MEET-3. +- No blocking product questions. Native platform packaging and speech-quality evaluation remain useful follow-up work. From 07ced656036b165422ce568fc34f65d6fb7d2803 Mon Sep 17 00:00:00 2001 From: Eliot Lim Date: Sun, 4 Oct 2026 07:13:38 +0800 Subject: [PATCH 3/4] fix(server): bound local transcription work (MEET-3) --- docs/local-transcription.md | 44 ++++++++++++ packages/server/src/ai/routes.ts | 23 +++++- packages/server/src/ai/service.ts | 10 ++- packages/server/src/ai/whisper.test.ts | 36 +++++++++- packages/server/src/ai/whisper.ts | 18 ++++- packages/server/src/transcription.test.ts | 79 ++++++++++++++++++++- packages/ui/src/components/AiSettings.tsx | 6 +- packages/ui/src/i18n/__tests__/i18n.test.ts | 11 +++ packages/ui/src/i18n/messages/de.ts | 12 ++++ packages/ui/src/i18n/messages/en.ts | 12 ++++ packages/ui/src/i18n/messages/ja.ts | 12 ++++ packages/ui/src/i18n/messages/zh.ts | 12 ++++ 12 files changed, 265 insertions(+), 10 deletions(-) create mode 100644 docs/local-transcription.md diff --git a/docs/local-transcription.md b/docs/local-transcription.md new file mode 100644 index 00000000..6accd3a2 --- /dev/null +++ b/docs/local-transcription.md @@ -0,0 +1,44 @@ +# Local transcription + +OpenBook transcribes recordings locally by default, without a cloud API key. The server uses the optional whisper.cpp `whisper-cli` executable and FFmpeg. Neither executable nor model weights are bundled with OpenBook; normal CI skips native inference explicitly. + +## Setup + +1. Install whisper.cpp and FFmpeg on the server host. Follow the [whisper.cpp build instructions](https://github.com/ggml-org/whisper.cpp#quick-start) or use your host package manager. +2. Put `whisper-cli` and `ffmpeg` on the server process's PATH. Alternatively, set executable paths before starting OpenBook: + + ```sh + export OPENBOOK_WHISPER_BIN=/absolute/path/to/whisper-cli + export OPENBOOK_FFMPEG_BIN=/absolute/path/to/ffmpeg + ``` + +3. In **Settings → AI**, select **Download Whisper base**. This uses the existing authenticated model download and progress flow. Downloads go to `OPENBOOK_MODELS_DIR` when set, otherwise the server data directory's `models` folder (or `~/.openbook/models` without a data directory). The completed model is discovered without restarting. +4. Check that Settings reports the runtime and model ready, then transcribe a recording. + +The default model is multilingual Whisper base (`ggml-base.bin`, approximately 142 MiB), with automatic language detection. It trades some accuracy on noisy speech, accents, and difficult multilingual recordings for a smaller download and lower memory use than larger models. Whisper downloads do not change the selected chat model. + +Explicit cloud transcription configuration takes precedence over local inference. Otherwise the server tries local transcription, then the existing mock fallback. Local transcription works even with chat disabled; explicitly disabling transcription still disables it. Missing executables or model weights produce an actionable error pointing to Settings → AI. + +## Processing and limits + +FFmpeg converts recordings to mono 16 kHz PCM WAV. Whisper loads the model for each job and returns text plus segments in seconds; `durationMs` is the rounded maximum segment endpoint in milliseconds. Each job uses a private temporary directory. Cancellation and server shutdown kill active child processes, wait for them to close, and remove scratch files. + +Each server permits at most **two local transcription jobs at once**, shared across all clients. There is no queue. Busy requests return HTTP **429** with `Retry-After: 5`. Permits remain held through temporary-file cleanup and are released on success, failure, or cancellation. + +The transcription route also allows **six local requests per socket IP per 60-second fixed window**. Excess requests return HTTP **429** with `Retry-After: 60`. Client-supplied forwarding headers do not change this key; clients behind a reverse proxy may share its socket IP budget. The limit applies only when the resolved backend is local; cloud and mock backends are unaffected. + +## Native smoke test + +Install the optional executables and download `ggml-base.bin` into your model directory, then run from the repository root: + +```sh +OPENBOOK_TEST_WHISPER=1 \ +OPENBOOK_MODELS_DIR=/absolute/path/to/models \ +OPENBOOK_WHISPER_BIN=/absolute/path/to/whisper-cli \ +OPENBOOK_FFMPEG_BIN=/absolute/path/to/ffmpeg \ +VITEST_MAX_WORKERS=1 \ +pnpm --filter @book.dev/server exec vitest run src/transcription.test.ts \ + -t 'native whisper' +``` + +This opt-in test synthesizes a small WAV and exercises the HTTP transcription route without cloud keys, asserting the result shape. It is a runtime integration check, not a speech-accuracy benchmark. Without `OPENBOOK_TEST_WHISPER=1`, the native test is explicitly skipped. The regular subprocess-fixture tests cover concurrency, cancellation, failures, timestamp conversion, and cleanup without installing native dependencies. diff --git a/packages/server/src/ai/routes.ts b/packages/server/src/ai/routes.ts index 3cb003ef..36250df9 100644 --- a/packages/server/src/ai/routes.ts +++ b/packages/server/src/ai/routes.ts @@ -6,11 +6,16 @@ import type {PageStore} from '../store'; import type {AppEnv} from '../appEnv'; import {isLocalInstanceOwner, requireAuthenticatedRead, requireCreate, requireInstanceAdmin, requireInstanceOwner} from '../access'; import {AgentRunner, type AgentMessage} from './agent'; -import {TranscriptionConfigError, type AiService} from './service'; +import {ModelDownloadConfigError, TranscriptionConfigError, type AiService} from './service'; import {McpConfigError, type ExternalAgentTool, type McpClientManager} from './mcpClients'; +import {FixedWindowLimiter, clientIpKey} from '../agentTokens'; +import {LocalTranscriptionBusyError} from './whisper'; import type {TokenUsage} from './providers'; import type {AiUsageLog, UsageKind} from './usage'; +export const LOCAL_TRANSCRIPTION_RATE_LIMIT = 6; +const LOCAL_TRANSCRIPTION_RATE_WINDOW_MS = 60_000; + /** * The `/api/ai/*` surface. Generation endpoints stream tokens as SSE * (`data: {"token": "..."}` frames, closed by `data: {"done": true}`); @@ -22,6 +27,7 @@ import type {AiUsageLog, UsageKind} from './usage'; * logging failure never breaks the request. */ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore, onPagesChanged?: () => Promise, aiUsage?: AiUsageLog, mcp?: McpClientManager): void { + const localTranscriptionLimiter = new FixedWindowLimiter(LOCAL_TRANSCRIPTION_RATE_LIMIT, LOCAL_TRANSCRIPTION_RATE_WINDOW_MS); /** * Log a single generate/complete usage row against the effective provider/model. * Best-effort (the logger swallows its own errors); does nothing without a logger @@ -195,12 +201,20 @@ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore try { const {engine, provider, model} = await ai.transcriptionBackend(); await requirePaidInferenceAccess(c, store, provider === 'openai-compat' ? 'openai' : 'off'); + if (provider === 'local' && localTranscriptionLimiter.exceeded(clientIpKey(c))) { + c.header('Retry-After', String(LOCAL_TRANSCRIPTION_RATE_WINDOW_MS / 1000)); + return c.json({error: 'Too many local transcription requests. Please retry shortly.'}, 429); + } const result = await engine.transcribe(asset.bytes, {mime: asset.mime, signal: c.req.raw.signal}); // Audio backends do not report token counts. Keep unknown cloud cost null. await aiUsage?.log({provider, model, kind: 'transcribe', principal, usage: {inputTokens: 0, outputTokens: 0}}); return c.json(result); } catch (err) { if (err instanceof HTTPException) throw err; + if (err instanceof LocalTranscriptionBusyError) { + c.header('Retry-After', String(err.retryAfterSeconds)); + return c.json({error: err.message}, 429); + } if (err instanceof TranscriptionConfigError) return c.json({error: err.message}, 400); return c.json({error: 'Transcription failed. Check the provider in Settings → AI and retry.'}, 502); } @@ -211,7 +225,12 @@ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore // surface) — only the trusted instance owner may supply it. await requireInstanceOwner(c, store); const {url} = (await c.req.json().catch(() => ({}))) as {url?: string}; - return c.json(await ai.startDownload(url)); + try { + return c.json(await ai.startDownload(url)); + } catch (err) { + if (err instanceof ModelDownloadConfigError) return c.json({error: err.message}, 400); + throw err; + } }); // The agent harness: runs the tool loop against the library and streams diff --git a/packages/server/src/ai/service.ts b/packages/server/src/ai/service.ts index de55c935..07aaf66a 100644 --- a/packages/server/src/ai/service.ts +++ b/packages/server/src/ai/service.ts @@ -97,6 +97,7 @@ interface DownloadState { } export class TranscriptionConfigError extends Error {} +export class ModelDownloadConfigError extends Error {} export class AiService { private config: AiConfig = DEFAULT_CONFIG; @@ -339,8 +340,13 @@ export class AiService { async startDownload(url = DEFAULT_MODEL_URL): Promise { await this.loadConfig(); if (this.download && !this.download.done && !this.download.error) return this.download; - const fileName = decodeURIComponent(new URL(url).pathname.split('/').pop() || 'model.gguf'); - if (fileName !== path.basename(fileName) || fileName === '.' || fileName === '..' || fileName.includes('\\')) throw new Error('Invalid model filename'); + let fileName: string; + try { + fileName = decodeURIComponent(new URL(url).pathname.split('/').pop() || 'model.gguf'); + } catch { + throw new ModelDownloadConfigError('Invalid model download URL or filename encoding.'); + } + if (fileName !== path.basename(fileName) || fileName === '.' || fileName === '..' || fileName.includes('\\')) throw new ModelDownloadConfigError('Invalid model filename'); mkdirSync(this.modelsDir, {recursive: true}); const dest = path.join(this.modelsDir, fileName); const state: DownloadState = {url, received: 0, total: null, done: false}; diff --git a/packages/server/src/ai/whisper.test.ts b/packages/server/src/ai/whisper.test.ts index d310b787..a9adece0 100644 --- a/packages/server/src/ai/whisper.test.ts +++ b/packages/server/src/ai/whisper.test.ts @@ -2,7 +2,7 @@ import {mkdtemp, readFile, readdir, rm, writeFile} from 'node:fs/promises'; import {tmpdir} from 'node:os'; import path from 'node:path'; import {afterEach, beforeEach, describe, expect, it} from 'vitest'; -import {LocalWhisper, parseWhisperOutput, runWhisperProcess, WHISPER_MODEL} from './whisper'; +import {LocalWhisper, LocalTranscriptionBusyError, parseWhisperOutput, runWhisperProcess, WHISPER_MODEL} from './whisper'; let dir: string; beforeEach(async () => { dir = await mkdtemp(path.join(tmpdir(), 'meet3-test-')); }); @@ -55,6 +55,40 @@ describe('optional local whisper runtime', () => { await expect(readdir(started.dir)).rejects.toThrow(); }); + it.each(['abort', 'failure'] as const)('limits jobs to two and releases permits after %s', async (outcome) => { + const mode = path.join(dir, 'mode'); + await writeFile(mode, 'wait'); + const ffmpeg = await script('ffmpeg', 'process.exit(0);'); + const whisper = await script('whisper', `const fs = require('node:fs'); const args = process.argv.slice(2); const output = args[args.indexOf('-of') + 1]; fs.writeFileSync(${JSON.stringify(dir)} + '/' + process.pid + '.started', ''); setInterval(() => { const mode = fs.readFileSync(${JSON.stringify(mode)}, 'utf8'); if (mode === 'fail') process.exit(1); if (mode === 'ok') { fs.writeFileSync(output + '.json', JSON.stringify({transcription: []})); process.exit(0); } }, 10);`); + const local = new LocalWhisper(dir, whisper, ffmpeg); + const controller = new AbortController(); + const jobs = Promise.allSettled([0, 1].map(() => local.transcribe(new Uint8Array([1]), {signal: controller.signal}))); + try { + await expect.poll(async () => (await readdir(dir)).filter((f) => f.endsWith('.started')).length).toBe(2); + await expect(local.transcribe(new Uint8Array([1]))).rejects.toBeInstanceOf(LocalTranscriptionBusyError); + if (outcome === 'abort') controller.abort(); + else await writeFile(mode, 'fail'); + const results = await jobs; + for (const result of results) { + expect(result.status).toBe('rejected'); + if (result.status === 'rejected') expect(result.reason.message).toContain(outcome === 'abort' ? 'abort' : 'Local audio processing failed'); + } + await writeFile(mode, 'ok'); + const recovered = await Promise.all([0, 1].map(() => local.transcribe(new Uint8Array([1])))); + expect(recovered).toEqual([{text: '', segments: [], durationMs: 0}, {text: '', segments: [], durationMs: 0}]); + } finally { + await local.dispose(); + await jobs; + } + }); + + it.each([1001, 1000.6])('keeps duration integer milliseconds without a seconds round-trip (%s)', (end) => { + expect(parseWhisperOutput({transcription: [ + {offsets: {from: 0, to: end}, text: 'First'}, + {offsets: {from: 0, to: 900}, text: 'Second'}, + ]}).durationMs).toBe(1001); + }); + it('rejects pre-aborted work, spawn failures, failed processes and malformed output', async () => { const controller = new AbortController(); controller.abort(); diff --git a/packages/server/src/ai/whisper.ts b/packages/server/src/ai/whisper.ts index bec7d535..989b0858 100644 --- a/packages/server/src/ai/whisper.ts +++ b/packages/server/src/ai/whisper.ts @@ -6,6 +6,15 @@ import path from 'node:path'; import type {AiStatus, AiTranscriptionResult} from '@book.dev/sdk'; import type {TranscriptionEngine, TranscribeOptions} from './providers'; +export const LOCAL_TRANSCRIPTION_MAX_JOBS = 2; + +export class LocalTranscriptionBusyError extends Error { + readonly retryAfterSeconds = 5; + constructor() { + super('Local transcription is busy. Please retry shortly.'); + } +} + export const WHISPER_MODEL = 'ggml-base.bin'; export const WHISPER_MODEL_URL = 'https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.bin'; @@ -47,6 +56,7 @@ export async function runWhisperProcess(command: string, args: string[], signal: export function parseWhisperOutput(value: unknown): AiTranscriptionResult { const data = value as {transcription?: {offsets?: {from?: number; to?: number}; text?: string}[]}; if (!data || !Array.isArray(data.transcription)) throw new Error('Invalid whisper output'); + let durationMs = 0; const segments = data.transcription.map((segment) => { const start = segment?.offsets?.from; const end = segment?.offsets?.to; @@ -54,9 +64,10 @@ export function parseWhisperOutput(value: unknown): AiTranscriptionResult { || typeof end !== 'number' || !Number.isFinite(end) || end < start || typeof segment.text !== 'string') { throw new Error('Invalid whisper segment'); } + durationMs = Math.max(durationMs, end); return {start: start / 1000, end: end / 1000, text: segment.text.trim()}; }); - return {text: segments.map((s) => s.text).join(' ').trim(), segments, durationMs: Math.max(0, ...segments.map((s) => s.end * 1000))}; + return {text: segments.map((s) => s.text).join(' ').trim(), segments, durationMs: Math.round(durationMs)}; } /** Optional whisper.cpp + FFmpeg runtime. Models live beside chat models, but @@ -91,6 +102,11 @@ export class LocalWhisper implements TranscriptionEngine { } transcribe(bytes: Uint8Array, opts: TranscribeOptions = {}): Promise { + if (opts.signal?.aborted) return Promise.reject(opts.signal.reason); + // This set is a non-queuing semaphore shared by all requests to this server. + // Reserve synchronously, before any await or temporary-file/process creation; + // keep the permit until perform's finally has removed the scratch directory. + if (this.active.size >= LOCAL_TRANSCRIPTION_MAX_JOBS) return Promise.reject(new LocalTranscriptionBusyError()); const controller = new AbortController(); const signal = opts.signal ? AbortSignal.any([opts.signal, controller.signal]) : controller.signal; this.active.add(controller); diff --git a/packages/server/src/transcription.test.ts b/packages/server/src/transcription.test.ts index b0216a34..1f5b5a92 100644 --- a/packages/server/src/transcription.test.ts +++ b/packages/server/src/transcription.test.ts @@ -12,7 +12,8 @@ import {MockEngine, OpenAiCompatEngine} from './ai/providers'; import {AiUsageLog} from './ai/usage'; import {LocalDataClient} from './localClient'; import {LocalWhisper, WHISPER_MODEL, WHISPER_MODEL_URL} from './ai/whisper'; -import {readFile, readdir} from 'node:fs/promises'; +import {LOCAL_TRANSCRIPTION_RATE_LIMIT} from './ai/routes'; +import {readFile, readdir, writeFile} from 'node:fs/promises'; let db: PgliteDb; let store: PageStore; @@ -151,6 +152,82 @@ describe('transcription contract', () => { await service.dispose(); }); + it('caps parallel local HTTP jobs across anonymous readers at two, returning 429 with Retry-After', async () => { + await store.updateInstanceConfig({ownerSubject: 'test#owner', guestAccess: 'read'}); + await store.setPageVisibility(pageId, 'public'); + const ffmpeg = join(dir, 'ffmpeg'); + const whisper = join(dir, 'whisper'); + const release = join(dir, 'release'); + await writeFile(ffmpeg, `#!${process.execPath}\nprocess.exit(0);`, {mode: 0o700}); + await writeFile(whisper, `#!${process.execPath}\nconst fs = require('node:fs'); const args = process.argv.slice(2); const output = args[args.indexOf('-of') + 1]; fs.writeFileSync(${JSON.stringify(dir)} + '/' + process.pid + '.started', ''); setInterval(() => { if (fs.existsSync(${JSON.stringify(release)})) { fs.writeFileSync(output + '.json', JSON.stringify({transcription: [{offsets: {from: 0, to: 1001}, text: 'Local'}]})); process.exit(0); } }, 10);`, {mode: 0o700}); + await writeFile(join(dir, WHISPER_MODEL), 'test model'); + const local = new LocalWhisper(dir, whisper, ffmpeg); + const service = new AiService(db, dir, () => local.resolve(), local); + const app = appWith(service); + const request = (ip: string) => app.request(API.aiTranscribe, { + method: 'POST', headers: {...headers, [FORWARDED_HEADER]: '1'}, body: JSON.stringify({assetId, pageId}), + }, {incoming: {socket: {remoteAddress: ip}}}); + const accepted = [request('192.0.2.1'), request('192.0.2.2')]; + try { + await expect.poll(async () => (await readdir(dir)).filter((f) => f.endsWith('.started')).length).toBe(2); + const busy = await request('192.0.2.3'); + expect(busy.status).toBe(429); + expect(busy.headers.get('Retry-After')).toBe('5'); + await writeFile(release, 'go'); + for (const response of await Promise.all(accepted)) { + expect(response.status).toBe(200); + expect(await response.json()).toMatchObject({text: 'Local', durationMs: 1001}); + } + expect((await request('192.0.2.3')).status).toBe(200); + } finally { + await service.dispose(); + await Promise.all(accepted); + } + }); + + it('rate-limits local requests per socket IP, ignores spoofed forwarding headers, and leaves mock/cloud untouched', async () => { + const engine = new MockEngine(); + const transcribe = vi.spyOn(engine, 'transcribe'); + let available = true; + const service = new AiService(db, dir, async () => available ? engine : null); + const app = appWith(service); + const request = (ip = '192.0.2.1', forwardedFor = '') => app.request(API.aiTranscribe, { + method: 'POST', headers: {...headers, [LOCAL_OWNER_HEADER]: secret, 'x-forwarded-for': forwardedFor}, body: JSON.stringify({assetId, pageId}), + }, {incoming: {socket: {remoteAddress: ip}}}); + for (let i = 0; i < LOCAL_TRANSCRIPTION_RATE_LIMIT; i++) expect((await request()).status).toBe(200); + const denied = await request('192.0.2.1', '198.51.100.1'); + expect(denied.status).toBe(429); + expect(denied.headers.get('Retry-After')).toBe('60'); + expect(transcribe).toHaveBeenCalledTimes(LOCAL_TRANSCRIPTION_RATE_LIMIT); + expect((await request('192.0.2.2')).status).toBe(200); + available = false; + await service.setConfig({provider: 'mock'}); + expect((await request()).status).toBe(200); + await service.setConfig({provider: 'off', transcription: {provider: 'openai-compat'}}); + vi.stubGlobal('fetch', vi.fn(async () => new Response(JSON.stringify({text: 'Cloud'})))); + expect((await request()).status).toBe(200); + available = true; + await service.setConfig({provider: 'off', transcription: {provider: 'local'}}); + vi.spyOn(Date, 'now').mockReturnValue(Date.now() + 60_001); + expect((await request()).status).toBe(200); + await service.dispose(); + }); + + it('returns 400 for malformed percent-encoding in model download filenames without fetching', async () => { + const fetchSpy = vi.spyOn(globalThis, 'fetch'); + const app = appWith(); + for (const filename of ['bad%.bin', 'bad%ZZ.bin', 'bad%E0%A4%A.bin']) { + const response = await app.request(API.aiModelDownload, { + method: 'POST', headers: {...headers, [LOCAL_OWNER_HEADER]: secret}, + body: JSON.stringify({url: `https://example.test/${filename}`}), + }); + expect(response.status).toBe(400); + expect(await response.json()).toEqual({error: 'Invalid model download URL or filename encoding.'}); + } + expect(fetchSpy).not.toHaveBeenCalled(); + expect((await ai.status()).download).toBeUndefined(); + }); + // Opt-in: OPENBOOK_TEST_WHISPER=1, OPENBOOK_MODELS_DIR containing ggml-base.bin, // whisper-cli + ffmpeg on PATH (or OPENBOOK_WHISPER_BIN / OPENBOOK_FFMPEG_BIN). it.skipIf(process.env.OPENBOOK_TEST_WHISPER !== '1')('native whisper: POST transcribes synthesized WAV with no cloud keys', async () => { diff --git a/packages/ui/src/components/AiSettings.tsx b/packages/ui/src/components/AiSettings.tsx index da02bb8a..88f3e421 100644 --- a/packages/ui/src/components/AiSettings.tsx +++ b/packages/ui/src/components/AiSettings.tsx @@ -274,10 +274,10 @@ export default function AiSettings() { return ( {status?.transcription && ( - -

{status.transcription.modelPresent ? 'Model downloaded.' : 'Model not downloaded.'} {status.transcription.ready ? 'Ready to transcribe.' : status.transcription.detail}

+ +

{t(status.transcription.modelPresent ? 'ai.transcription.modelPresent' : 'ai.transcription.modelAbsent')} {t(status.transcription.ready ? 'ai.transcription.ready' : status.transcription.runtimeAvailable ? 'ai.transcription.modelMissing' : 'ai.transcription.runtimeMissing')}

{download?.url === status.transcription.downloadUrl && download.error &&

{download.error}

}
diff --git a/packages/ui/src/i18n/__tests__/i18n.test.ts b/packages/ui/src/i18n/__tests__/i18n.test.ts index 49e49b59..5386fac3 100644 --- a/packages/ui/src/i18n/__tests__/i18n.test.ts +++ b/packages/ui/src/i18n/__tests__/i18n.test.ts @@ -18,6 +18,17 @@ describe('t', () => { expect(t('common.cancel')).toBe('取消'); }); + it('ships local transcription copy and progress in all four locales', () => { + for (const catalog of [en, de, ja, zh]) { + expect(Object.keys(catalog.ai?.transcription ?? {}).sort()).toEqual(Object.keys(en.ai.transcription).sort()); + expect(Object.values(catalog.ai?.transcription ?? {}).every((value) => typeof value === 'string' && value.length > 0)).toBe(true); + } + for (const locale of ['en', 'de', 'ja', 'zh'] as const) { + setLocale(locale); + expect(t('ai.transcription.downloadingProgress', {progress: 42})).toContain('42%'); + } + }); + it('interpolates {var} placeholders', () => { setLocale('en'); expect(t('mention.create', {name: 'Roadmap'})).toBe('Create subpage “Roadmap”'); diff --git a/packages/ui/src/i18n/messages/de.ts b/packages/ui/src/i18n/messages/de.ts index 789dca6a..3800b83b 100644 --- a/packages/ui/src/i18n/messages/de.ts +++ b/packages/ui/src/i18n/messages/de.ts @@ -532,6 +532,18 @@ export const de: PartialMessages = { localNetwork: 'Lokales Netzwerk', }, ai: { + transcription: { + title: 'Lokale Transkription', + description: 'Aufnahmen werden standardmäßig lokal mit Whisper transkribiert, ohne Cloud-Schlüssel. Whisper base ist mehrsprachig (~142 MiB).', + modelPresent: 'Modell heruntergeladen.', + modelAbsent: 'Modell nicht heruntergeladen.', + ready: 'Bereit zur Transkription.', + runtimeMissing: 'Installiere whisper-cli und FFmpeg auf dem Server, um die lokale Transkription zu aktivieren.', + modelMissing: 'Lade unten das Modell herunter, um die lokale Transkription zu aktivieren.', + download: 'Whisper base herunterladen', + downloading: 'Wird heruntergeladen…', + downloadingProgress: 'Download: {progress}%', + }, title: 'KI', description: 'Ein optionales lokales Modell ermöglicht Notizsuche, Aufgabenplanung und Textvervollständigung. Alles läuft auf deinem Gerät.', providerLabel: 'Engine', diff --git a/packages/ui/src/i18n/messages/en.ts b/packages/ui/src/i18n/messages/en.ts index b2ed5631..4b49a7b4 100644 --- a/packages/ui/src/i18n/messages/en.ts +++ b/packages/ui/src/i18n/messages/en.ts @@ -846,6 +846,18 @@ export const en = { removeBody: 'This only removes it from this device’s library list. The server and its data are untouched.', }, ai: { + transcription: { + title: 'Local transcription', + description: 'Recordings use Whisper locally by default, with no cloud key. Whisper base is multilingual (~142 MiB).', + modelPresent: 'Model downloaded.', + modelAbsent: 'Model not downloaded.', + ready: 'Ready to transcribe.', + runtimeMissing: 'Install whisper-cli and FFmpeg on the server to enable local transcription.', + modelMissing: 'Download the model below to enable local transcription.', + download: 'Download Whisper base', + downloading: 'Downloading…', + downloadingProgress: 'Downloading {progress}%', + }, title: 'AI', description: 'An optional model powers note search, task breakdown, and writing help. Run it locally — nothing leaves your machine — or connect the Claude API.', providerLabel: 'Engine', diff --git a/packages/ui/src/i18n/messages/ja.ts b/packages/ui/src/i18n/messages/ja.ts index 4628f2d4..10efb8b2 100644 --- a/packages/ui/src/i18n/messages/ja.ts +++ b/packages/ui/src/i18n/messages/ja.ts @@ -527,6 +527,18 @@ export const ja: PartialMessages = { localNetwork: 'ローカルネットワーク', }, ai: { + transcription: { + title: 'ローカル文字起こし', + description: '録音は標準でWhisperを使ってローカルで文字起こしされます。クラウドのAPIキーは不要です。Whisper baseは多言語対応です(約142 MiB)。', + modelPresent: 'モデルをダウンロード済みです。', + modelAbsent: 'モデルが未ダウンロードです。', + ready: '文字起こしの準備ができています。', + runtimeMissing: 'ローカル文字起こしを有効にするには、サーバーにwhisper-cliとFFmpegをインストールしてください。', + modelMissing: 'ローカル文字起こしを有効にするには、以下のモデルをダウンロードしてください。', + download: 'Whisper baseをダウンロード', + downloading: 'ダウンロード中…', + downloadingProgress: 'ダウンロード中 {progress}%', + }, title: 'AI', description: 'オプションのローカルモデルで、ノート検索・タスク分解・文書補完ができます。すべて端末内で動作します。', providerLabel: 'エンジン', diff --git a/packages/ui/src/i18n/messages/zh.ts b/packages/ui/src/i18n/messages/zh.ts index c31cb9fa..89773e43 100644 --- a/packages/ui/src/i18n/messages/zh.ts +++ b/packages/ui/src/i18n/messages/zh.ts @@ -526,6 +526,18 @@ export const zh: PartialMessages = { localNetwork: '局域网', }, ai: { + transcription: { + title: '本地转录', + description: '录音默认使用 Whisper 在本地转录,无需云端密钥。Whisper base 支持多种语言(约142 MiB)。', + modelPresent: '模型已下载。', + modelAbsent: '模型尚未下载。', + ready: '已准备好转录。', + runtimeMissing: '请在服务器上安装 whisper-cli 和 FFmpeg 以启用本地转录。', + modelMissing: '请下载下方模型以启用本地转录。', + download: '下载 Whisper base', + downloading: '正在下载…', + downloadingProgress: '正在下载 {progress}%', + }, title: 'AI', description: '可选的本地模型支持笔记搜索、任务分解和文档补全。一切都在你的设备上运行。', providerLabel: '引擎', From 0a49bb1686117af4345d5200a818af1c542cb890 Mon Sep 17 00:00:00 2001 From: Eliot Lim Date: Sun, 4 Oct 2026 07:14:33 +0800 Subject: [PATCH 4/4] chore: drop worker artifacts (MEET-3) --- _brief-meet3.md | 29 ---------------------- _report-meet3.md | 62 ------------------------------------------------ 2 files changed, 91 deletions(-) delete mode 100644 _brief-meet3.md delete mode 100644 _report-meet3.md diff --git a/_brief-meet3.md b/_brief-meet3.md deleted file mode 100644 index f9a5881b..00000000 --- a/_brief-meet3.md +++ /dev/null @@ -1,29 +0,0 @@ -# MEET-3 — Local whisper engine (default transcription path) - -You are Finley, an OpenBook Worker agent. Work in THIS worktree (`/Users/eliot/Workspaces/OpenBook-wt-meet-2`). Do NOT push. Conventional commits (`feat(server): … (MEET-3)`), committed incrementally. Disk headroom is ~6 GiB — prefer a SMALL whisper model for tests/default; clean temp artifacts. - -## Branch setup -`git checkout -b feat/meet-3-local-whisper` from current HEAD (feat/meet-2-transcribe @ bbccdbc1). This stacks on MEET-2; merge order to main is MEET-2 first. - -## Task -Implement the LOCAL transcription backend so transcription works with zero cloud keys — this is the product DEFAULT (owner directive). MEET-2 built the exact seam for you; its contract: - -- Implement `TranscriptionEngine` from `packages/server/src/ai/providers.ts`: `transcribe(bytes: Uint8Array, opts?: {filename?, mime?, signal?}): Promise` ({text, segments?[{start,end,text}] seconds, durationMs}). Respect AbortSignal. -- Inject via the optional third `AiService` constructor argument `() => Promise` (resolve/start the managed backend lazily; return null when unavailable). MEET-3 owns process/model lifecycle; wire the resolver at AiService construction in server startup. -- Resolution order already implemented by MEET-2: explicit cloud > your local resolver > mock > actionable 400. Local is ungated (no paid gate). Usage rows log kind 'transcribe' provider 'local', cost 0. - -## Implementation guidance -- Follow the `LlamaEngine` optional-native-dep pattern (providers.ts ~:468, node-llama-cpp): whisper binding as an OPTIONAL dependency — evaluate smart-whisper vs whisper.cpp node addons vs spawning a whisper.cpp/whisper-server binary; pick fewest native-build headaches across mac/linux CI and justify in your report. An OpenAI-compatible local server (reusing `new OpenAiCompatEngine(baseUrl, model)` against a managed localhost process) is also acceptable if it's the most robust path. -- Model acquisition reuses the existing model-download flow (ai/service.ts startDownload pattern, OPENBOOK_MODELS_DIR, server.ts ~:437). Default model: whisper base or small multilingual (ggml); state the size/quality trade-off. Do NOT bundle model bytes in git. -- Absence of the native dep / model must degrade to a clear error pointing at Settings → AI model download — never a crash. CI (`pnpm verify`) must be green WITHOUT the native dep installed (mock/skip pattern like llama). -- Settings → AI: expose local transcription state (model present/absent, download affordance) through the existing aiStatus/config surfaces; keep UI changes minimal — MEET-5's Settings polish is out of scope. - -## Acceptance (each maps to a test) -1. With the binding available + a model present and no cloud config, POST /api/ai/transcribe returns real text+segments for a small fixture (check in a fixture only if <100 KB; else synthesize audio in the test and assert non-error shape). Gate this test behind an env flag or dep-presence check so CI without the dep skips it EXPLICITLY (visible skip, not silent). -2. Resolver returns null when dep/model missing → route falls through per MEET-2 order; actionable 400 text mentions Settings → AI. -3. AbortSignal cancels an in-flight local transcription. -4. Usage row: provider 'local', cost 0. -5. `pnpm verify` green FOREGROUND (VITEST_MAX_WORKERS=1 if the money.test.ts load flake appears — do not modify that test), WITHOUT the native dep in the default run. - -## Done -All committed, not pushed. Write `_report-meet3.md`: outcome first, head sha, dep choice + rationale, model default + trade-off, criterion→test map, deviations, open questions. Delete nothing from existing tests. Never poll external state; wedged after a real attempt → commit + report. diff --git a/_report-meet3.md b/_report-meet3.md deleted file mode 100644 index 149f1db7..00000000 --- a/_report-meet3.md +++ /dev/null @@ -1,62 +0,0 @@ -# MEET-3 — Finley report - -Implemented the default local transcription path, managed process cleanup, model download/status, and the minimal Settings → AI affordance. No cloud key is required. Full foreground `pnpm verify` passed, as did native HTTP and real-speech smoke checks. Nothing was pushed. - -- Branch: `feat/meet-3-local-whisper`, stacked on `feat/meet-2-transcribe` at `bbccdbc1`. -- Implementation HEAD: `0a2adfa9f6d472941b7494a03c8a34a7a0901b18`. The subsequent report-only commit records verification; its SHA is supplied in the worker response. -- Merge MEET-2 before MEET-3. - -## Runtime choice and model - -Chose the optional **whisper.cpp `whisper-cli` system executable**, plus **FFmpeg** for browser WebM/MP4/Ogg and other audio inputs. No new npm native addon or mandatory install/build hook is introduced. Compared with [smart-whisper's native Node addon](https://github.com/JacobLinCool/smart-whisper), the subprocess approach avoids Node ABI coupling and addon build failures in macOS/Linux CI. Compared with a persistent whisper-server, it needs no port allocation, readiness polling, or resident model memory. The cost is loading the model for each recording and requiring host-installed binaries. - -The [upstream whisper.cpp CLI](https://github.com/ggml-org/whisper.cpp/tree/master/examples/cli) supports JSON output. Its millisecond offsets become API segments in seconds; their final endpoint supplies `durationMs`, following MEET-2's segment-derived timing fallback. FFmpeg normalizes input to mono 16 kHz PCM WAV. The server spawns commands directly without a shell, uses private temporary directories, ignores caller filenames, kills work on abort/shutdown, and removes scratch files after children close. - -Default: **multilingual Whisper base**, `ggml-base.bin`, approximately **142 MiB**. It is a smaller download and uses less memory than small/medium models, with lower accuracy on noisy speech, accents, and difficult multilingual recordings. Language detection is automatic. Model bytes are never committed. - -Setup: - -1. Install whisper.cpp (`whisper-cli`) and FFmpeg on the server host using the [upstream build instructions](https://github.com/ggml-org/whisper.cpp#quick-start) or the host package manager. -2. Make both executables available on the server's PATH. Alternatively set `OPENBOOK_WHISPER_BIN` and `OPENBOOK_FFMPEG_BIN` to executable paths before starting OpenBook. -3. In Settings → AI, select **Download Whisper base**. The existing authenticated model-download flow downloads into `OPENBOOK_MODELS_DIR`, or the server's existing default models directory. The local resolver discovers the completed model without restart. -4. Local is the transcription default even when chat is off. Explicit cloud configuration still takes precedence, then local, then the existing mock fallback. Explicit transcription off stays off. - -## Criterion → test map - -| Acceptance | Evidence | -| --- | --- | -| 1. Native local POST, no cloud keys, result with segments | `transcription.test.ts`: `native whisper: POST transcribes synthesized WAV with no cloud keys`. Explicitly gated by `OPENBOOK_TEST_WHISPER=1`; synthesizes a 32 KB WAV in memory and asserts non-error result shape. **Passed in a separate opt-in native run** with temporary binaries/model; default verification still skips it explicitly. An additional real-speech sample returned nonempty text and timed segments. | -| 2. Missing runtime/model → null, fallback, actionable 400 | `ai/whisper.test.ts`: missing model/executable and discovery-without-restart test. `transcription.test.ts`: managed resolver fallback and aiStatus test; existing off/unavailable/cloud-precedence tests retained. | -| 3. Abort cancels in-flight local inference | `ai/whisper.test.ts`: actual child-process cancellation and disposal tests; both verify the child is gone and scratch directory removed. Also tests pre-aborted work and process failures. | -| 4. Usage provider local, cost 0 | `transcription.test.ts`: local usage/download test asserts `provider: local`, `model: ggml-base.bin`, `kind: transcribe`, `cost: 0`. | -| 5. Foreground verify, no native dependency | Verification results below. | - -Additional tests cover exact input-byte handling, timestamp normalization, malformed JSON shape, empty transcription, temporary-file cleanup, owner-route model download, and preventing Whisper downloads from becoming the llama chat model. No existing tests were deleted. - -Native smoke command after installing the optional binaries/model: - -```sh -OPENBOOK_TEST_WHISPER=1 OPENBOOK_MODELS_DIR=/absolute/path/to/models \ - pnpm --filter @book.dev/server exec vitest run src/transcription.test.ts -``` - -## Verification - -- Focused tests: **17 passed, 1 explicitly skipped** across the transcription and whisper suites. -- Implementation commit hooks: ESLint, SDK/server/UI typechecks, and conventional commit validation passed. -- Opt-in native HTTP smoke: **1 passed** (12 unrelated tests skipped by the name filter), 11.11 seconds. Temporary CPU-only whisper.cpp build at `60c0be6ac8fa71b1a2ae2dd938a31a34a508e774`, FFmpeg reporting version 6.0, and multilingual base on macOS arm64. -- Real-speech smoke: the upstream temporary `samples/jfk.wav` produced the expected “ask not” text, a segment spanning 0–10.5 seconds, and `durationMs: 10500`. No sample/model bytes were added to git. -- Native build tools, binaries, source, model, and scratch files were removed after the smoke runs. The isolated downloads/build occupied roughly 400 MiB and never entered the server's default PATH or model directory. -- First foreground `pnpm verify` inside the macOS sandbox: all earlier stages passed; server tests finished with 98 files / 1,342 tests passing, 7 skipped, and one mirror integration failure accompanied by 14 `EMFILE` filesystem-watch errors. The file-descriptor soft limit was already 1,048,575; a standalone `fs.watch` on a new empty temporary directory also emitted `EMFILE` inside the sandbox, confirming the environment restriction. -- The unchanged mirror integration suite passed all 3 tests outside the sandbox (14.43 seconds). No existing tests or resource limits were changed. -- Full foreground `pnpm verify` rerun outside the sandbox: **PASSED, exit 0**, with no optional Whisper runtime/model installed in the default environment. Builds, generated-file checks, typechecks, lint, all package tests, and end-to-end checks passed. SDK: 546 tests; UI: 2,319; app: 7; server: 99 files, 1,343 tests passed and 7 explicitly skipped. Server end-to-end: 256 checks; MCP end-to-end: 70 checks. The server unit suite took 3,397.77 seconds. - -Local verification artifacts: [verify-green.log](/Users/eliot/.bb/thread-storage/meet3-verification-gl8tvey3/verify-green.log), [verify-sandbox-failure.log](/Users/eliot/.bb/thread-storage/meet3-verification-gl8tvey3/verify-sandbox-failure.log), [native-http.log](/Users/eliot/.bb/thread-storage/meet3-verification-gl8tvey3/native-http.log), [native-speech.log](/Users/eliot/.bb/thread-storage/meet3-verification-gl8tvey3/native-speech.log). - -## Deviations and open questions - -- Used the brief's permitted optional executable alternative, not an npm `optionalDependencies` entry. CI requires neither whisper.cpp nor FFmpeg. Product packaging/distribution of those binaries remains a follow-up; Settings explicitly explains missing runtime requirements. -- The manager supplies status/disposal through a fourth optional `AiService` constructor argument, preserving the existing third-argument resolver contract and existing callers. -- No native runtime/model was installed into the workspace or globally. The default suite explicitly skips native inference; the separate opt-in run passed using temporary executables/model. Linux and GPU backends were not exercised here. -- Model downloads reuse the existing single-download slot and progress surface. The local model is fixed to base for this milestone; selecting larger local models and broader Settings polish remain outside MEET-3. -- No blocking product questions. Native platform packaging and speech-quality evaluation remain useful follow-up work.