diff --git a/docs/local-transcription.md b/docs/local-transcription.md new file mode 100644 index 00000000..6accd3a2 --- /dev/null +++ b/docs/local-transcription.md @@ -0,0 +1,44 @@ +# Local transcription + +OpenBook transcribes recordings locally by default, without a cloud API key. The server uses the optional whisper.cpp `whisper-cli` executable and FFmpeg. Neither executable nor model weights are bundled with OpenBook; normal CI skips native inference explicitly. + +## Setup + +1. Install whisper.cpp and FFmpeg on the server host. Follow the [whisper.cpp build instructions](https://github.com/ggml-org/whisper.cpp#quick-start) or use your host package manager. +2. Put `whisper-cli` and `ffmpeg` on the server process's PATH. Alternatively, set executable paths before starting OpenBook: + + ```sh + export OPENBOOK_WHISPER_BIN=/absolute/path/to/whisper-cli + export OPENBOOK_FFMPEG_BIN=/absolute/path/to/ffmpeg + ``` + +3. In **Settings → AI**, select **Download Whisper base**. This uses the existing authenticated model download and progress flow. Downloads go to `OPENBOOK_MODELS_DIR` when set, otherwise the server data directory's `models` folder (or `~/.openbook/models` without a data directory). The completed model is discovered without restarting. +4. Check that Settings reports the runtime and model ready, then transcribe a recording. + +The default model is multilingual Whisper base (`ggml-base.bin`, approximately 142 MiB), with automatic language detection. It trades some accuracy on noisy speech, accents, and difficult multilingual recordings for a smaller download and lower memory use than larger models. Whisper downloads do not change the selected chat model. + +Explicit cloud transcription configuration takes precedence over local inference. Otherwise the server tries local transcription, then the existing mock fallback. Local transcription works even with chat disabled; explicitly disabling transcription still disables it. Missing executables or model weights produce an actionable error pointing to Settings → AI. + +## Processing and limits + +FFmpeg converts recordings to mono 16 kHz PCM WAV. Whisper loads the model for each job and returns text plus segments in seconds; `durationMs` is the rounded maximum segment endpoint in milliseconds. Each job uses a private temporary directory. Cancellation and server shutdown kill active child processes, wait for them to close, and remove scratch files. + +Each server permits at most **two local transcription jobs at once**, shared across all clients. There is no queue. Busy requests return HTTP **429** with `Retry-After: 5`. Permits remain held through temporary-file cleanup and are released on success, failure, or cancellation. + +The transcription route also allows **six local requests per socket IP per 60-second fixed window**. Excess requests return HTTP **429** with `Retry-After: 60`. Client-supplied forwarding headers do not change this key; clients behind a reverse proxy may share its socket IP budget. The limit applies only when the resolved backend is local; cloud and mock backends are unaffected. + +## Native smoke test + +Install the optional executables and download `ggml-base.bin` into your model directory, then run from the repository root: + +```sh +OPENBOOK_TEST_WHISPER=1 \ +OPENBOOK_MODELS_DIR=/absolute/path/to/models \ +OPENBOOK_WHISPER_BIN=/absolute/path/to/whisper-cli \ +OPENBOOK_FFMPEG_BIN=/absolute/path/to/ffmpeg \ +VITEST_MAX_WORKERS=1 \ +pnpm --filter @book.dev/server exec vitest run src/transcription.test.ts \ + -t 'native whisper' +``` + +This opt-in test synthesizes a small WAV and exercises the HTTP transcription route without cloud keys, asserting the result shape. It is a runtime integration check, not a speech-accuracy benchmark. Without `OPENBOOK_TEST_WHISPER=1`, the native test is explicitly skipped. The regular subprocess-fixture tests cover concurrency, cancellation, failures, timestamp conversion, and cleanup without installing native dependencies. diff --git a/packages/sdk/src/ai.ts b/packages/sdk/src/ai.ts index 5212ad79..1a0662e6 100644 --- a/packages/sdk/src/ai.ts +++ b/packages/sdk/src/ai.ts @@ -212,6 +212,8 @@ export interface AiUsageResponse { } export interface AiStatus { + /** Local audio capability is independent of the chat provider. */ + transcription?: {model: string; modelPresent: boolean; runtimeAvailable: boolean; ready: boolean; downloadUrl: string; detail?: string}; config: AiConfig; /** The engine can generate text right now. */ ready: boolean; diff --git a/packages/server/src/ai/routes.ts b/packages/server/src/ai/routes.ts index 3cb003ef..36250df9 100644 --- a/packages/server/src/ai/routes.ts +++ b/packages/server/src/ai/routes.ts @@ -6,11 +6,16 @@ import type {PageStore} from '../store'; import type {AppEnv} from '../appEnv'; import {isLocalInstanceOwner, requireAuthenticatedRead, requireCreate, requireInstanceAdmin, requireInstanceOwner} from '../access'; import {AgentRunner, type AgentMessage} from './agent'; -import {TranscriptionConfigError, type AiService} from './service'; +import {ModelDownloadConfigError, TranscriptionConfigError, type AiService} from './service'; import {McpConfigError, type ExternalAgentTool, type McpClientManager} from './mcpClients'; +import {FixedWindowLimiter, clientIpKey} from '../agentTokens'; +import {LocalTranscriptionBusyError} from './whisper'; import type {TokenUsage} from './providers'; import type {AiUsageLog, UsageKind} from './usage'; +export const LOCAL_TRANSCRIPTION_RATE_LIMIT = 6; +const LOCAL_TRANSCRIPTION_RATE_WINDOW_MS = 60_000; + /** * The `/api/ai/*` surface. Generation endpoints stream tokens as SSE * (`data: {"token": "..."}` frames, closed by `data: {"done": true}`); @@ -22,6 +27,7 @@ import type {AiUsageLog, UsageKind} from './usage'; * logging failure never breaks the request. */ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore, onPagesChanged?: () => Promise, aiUsage?: AiUsageLog, mcp?: McpClientManager): void { + const localTranscriptionLimiter = new FixedWindowLimiter(LOCAL_TRANSCRIPTION_RATE_LIMIT, LOCAL_TRANSCRIPTION_RATE_WINDOW_MS); /** * Log a single generate/complete usage row against the effective provider/model. * Best-effort (the logger swallows its own errors); does nothing without a logger @@ -195,12 +201,20 @@ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore try { const {engine, provider, model} = await ai.transcriptionBackend(); await requirePaidInferenceAccess(c, store, provider === 'openai-compat' ? 'openai' : 'off'); + if (provider === 'local' && localTranscriptionLimiter.exceeded(clientIpKey(c))) { + c.header('Retry-After', String(LOCAL_TRANSCRIPTION_RATE_WINDOW_MS / 1000)); + return c.json({error: 'Too many local transcription requests. Please retry shortly.'}, 429); + } const result = await engine.transcribe(asset.bytes, {mime: asset.mime, signal: c.req.raw.signal}); // Audio backends do not report token counts. Keep unknown cloud cost null. await aiUsage?.log({provider, model, kind: 'transcribe', principal, usage: {inputTokens: 0, outputTokens: 0}}); return c.json(result); } catch (err) { if (err instanceof HTTPException) throw err; + if (err instanceof LocalTranscriptionBusyError) { + c.header('Retry-After', String(err.retryAfterSeconds)); + return c.json({error: err.message}, 429); + } if (err instanceof TranscriptionConfigError) return c.json({error: err.message}, 400); return c.json({error: 'Transcription failed. Check the provider in Settings → AI and retry.'}, 502); } @@ -211,7 +225,12 @@ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore // surface) — only the trusted instance owner may supply it. await requireInstanceOwner(c, store); const {url} = (await c.req.json().catch(() => ({}))) as {url?: string}; - return c.json(await ai.startDownload(url)); + try { + return c.json(await ai.startDownload(url)); + } catch (err) { + if (err instanceof ModelDownloadConfigError) return c.json({error: err.message}, 400); + throw err; + } }); // The agent harness: runs the tool loop against the library and streams diff --git a/packages/server/src/ai/service.ts b/packages/server/src/ai/service.ts index f03187b1..07aaf66a 100644 --- a/packages/server/src/ai/service.ts +++ b/packages/server/src/ai/service.ts @@ -5,6 +5,7 @@ import {providerSettings, type AiConfig, type AiProvider, type AiProviderSetting import type {Db} from '../db'; import {createEngine, MockEngine, OpenAiCompatEngine, type TranscriptionEngine, type AiEngine, type GenerateOptions} from './providers'; import {assembleSearchResults, bm25Scores, buildIndex, cosine, pageRowsToDocs, parseTaskList, type Bm25Index} from './search'; +import {WHISPER_MODEL, WHISPER_MODEL_URL} from './whisper'; import {SkillStore} from './skills'; /** @@ -96,6 +97,7 @@ interface DownloadState { } export class TranscriptionConfigError extends Error {} +export class ModelDownloadConfigError extends Error {} export class AiService { private config: AiConfig = DEFAULT_CONFIG; @@ -114,6 +116,7 @@ export class AiService { private readonly modelsDir: string, /** MEET-3: lazily resolve the managed local audio backend. */ private readonly localTranscription?: () => Promise, + private readonly localLifecycle?: {status(): Promise>; dispose(): Promise}, ) { this.skills = new SkillStore(db); } @@ -162,9 +165,9 @@ export class AiService { return {engine: new OpenAiCompatEngine(audio.baseUrl?.trim() || 'https://api.openai.com', audio.model?.trim() || 'whisper-1', 'openai', audio.apiKey), provider: 'openai-compat', model: audio.model?.trim() || 'whisper-1'}; } const local = await this.localTranscription?.(); - if (local) return {engine: local, provider: 'local', model: audio?.model ?? 'local'}; + if (local) return {engine: local, provider: 'local', model: WHISPER_MODEL}; if (config.provider === 'mock') return {engine: new MockEngine(), provider: 'mock', model: 'mock'}; - throw new TranscriptionConfigError('Local transcription is unavailable. Configure transcription in Settings → AI.'); + throw new TranscriptionConfigError('Local transcription is unavailable. Download the Whisper model and check the local runtime in Settings → AI.'); } async status(): Promise { @@ -183,6 +186,7 @@ export class AiService { } return { config: this.config, + transcription: await this.localLifecycle?.status(), ready, embeddings, detail, @@ -336,7 +340,13 @@ export class AiService { async startDownload(url = DEFAULT_MODEL_URL): Promise { await this.loadConfig(); if (this.download && !this.download.done && !this.download.error) return this.download; - const fileName = decodeURIComponent(new URL(url).pathname.split('/').pop() || 'model.gguf'); + let fileName: string; + try { + fileName = decodeURIComponent(new URL(url).pathname.split('/').pop() || 'model.gguf'); + } catch { + throw new ModelDownloadConfigError('Invalid model download URL or filename encoding.'); + } + if (fileName !== path.basename(fileName) || fileName === '.' || fileName === '..' || fileName.includes('\\')) throw new ModelDownloadConfigError('Invalid model filename'); mkdirSync(this.modelsDir, {recursive: true}); const dest = path.join(this.modelsDir, fileName); const state: DownloadState = {url, received: 0, total: null, done: false}; @@ -367,7 +377,7 @@ export class AiService { state.done = true; } // Auto-select the downloaded model for the llama provider. - if (this.config.provider === 'llama' && !this.config.model) { + if (url !== WHISPER_MODEL_URL && this.config.provider === 'llama' && !this.config.model) { await this.setConfig({...this.config, model: fileName}); } } catch (err) { @@ -381,5 +391,6 @@ export class AiService { async dispose(): Promise { await this.engine?.dispose().catch(() => undefined); + await this.localLifecycle?.dispose(); } } diff --git a/packages/server/src/ai/whisper.test.ts b/packages/server/src/ai/whisper.test.ts new file mode 100644 index 00000000..a9adece0 --- /dev/null +++ b/packages/server/src/ai/whisper.test.ts @@ -0,0 +1,104 @@ +import {mkdtemp, readFile, readdir, rm, writeFile} from 'node:fs/promises'; +import {tmpdir} from 'node:os'; +import path from 'node:path'; +import {afterEach, beforeEach, describe, expect, it} from 'vitest'; +import {LocalWhisper, LocalTranscriptionBusyError, parseWhisperOutput, runWhisperProcess, WHISPER_MODEL} from './whisper'; + +let dir: string; +beforeEach(async () => { dir = await mkdtemp(path.join(tmpdir(), 'meet3-test-')); }); +afterEach(async () => { await rm(dir, {recursive: true, force: true}); }); + +async function script(name: string, source: string): Promise { + const file = path.join(dir, name); + await writeFile(file, `#!${process.execPath}\n${source}`, {mode: 0o700}); + return file; +} + +describe('optional local whisper runtime', () => { + it('returns null for missing model or executables, and discovers a model without restart', async () => { + const local = new LocalWhisper(dir, process.execPath, process.execPath); + expect(await local.resolve()).toBeNull(); + expect(await local.status()).toMatchObject({modelPresent: false, runtimeAvailable: true, ready: false}); + await writeFile(path.join(dir, WHISPER_MODEL), 'test model'); + expect(await local.resolve()).toBe(local); + const missing = new LocalWhisper(dir, path.join(dir, 'missing'), process.execPath); + expect(await missing.resolve()).toBeNull(); + expect(await missing.status()).toMatchObject({modelPresent: true, runtimeAvailable: false, detail: expect.stringContaining('Settings → AI')}); + await local.dispose(); + expect(await local.resolve()).toBeNull(); + }); + + it('decodes exact input bytes, converts millisecond offsets, and cleans scratch files', async () => { + const marker = path.join(dir, 'scratch'); + const ffmpeg = await script('ffmpeg', 'const fs = require(\'node:fs\'); const args = process.argv.slice(2); const input = args[args.indexOf(\'-i\') + 1]; if (fs.readFileSync(input).toString() !== \'recording\') process.exit(2); fs.writeFileSync(args.at(-1), \'wav\');'); + const whisper = await script('whisper', `const fs = require('node:fs'); const args = process.argv.slice(2); const output = args[args.indexOf('-of') + 1]; fs.writeFileSync(${JSON.stringify(marker)}, require('node:path').dirname(output)); fs.writeFileSync(output + '.json', JSON.stringify({transcription: [{offsets: {from: 250, to: 1500}, text: ' Hello world.'}]}));`); + const local = new LocalWhisper(dir, whisper, ffmpeg); + expect(await local.transcribe(new TextEncoder().encode('recording'), {filename: '../../escape.webm', mime: 'audio/webm'})) + .toEqual({text: 'Hello world.', segments: [{start: 0.25, end: 1.5, text: 'Hello world.'}], durationMs: 1500}); + await expect(readdir(await readFile(marker, 'utf8'))).rejects.toThrow(); + }); + + it.each(['abort', 'dispose'] as const)('%s kills in-flight inference and removes scratch files', async (action) => { + const marker = path.join(dir, 'started'); + const ffmpeg = await script('ffmpeg', 'process.exit(0);'); + const whisper = await script('whisper', `const fs = require('node:fs'); const args = process.argv.slice(2); fs.writeFileSync(${JSON.stringify(marker)}, JSON.stringify({pid: process.pid, dir: require('node:path').dirname(args[args.indexOf('-of') + 1])})); setInterval(() => {}, 1000);`); + const local = new LocalWhisper(dir, whisper, ffmpeg); + const controller = new AbortController(); + const promise = local.transcribe(new Uint8Array([1]), {signal: controller.signal}); + const rejected = expect(promise).rejects.toMatchObject({name: 'AbortError'}); + await expect.poll(async () => readFile(marker, 'utf8').catch(() => '')).not.toBe(''); + const started = JSON.parse(await readFile(marker, 'utf8')) as {pid: number; dir: string}; + if (action === 'abort') controller.abort(); + else await local.dispose(); + await rejected; + expect(() => process.kill(started.pid, 0)).toThrow(); + await expect(readdir(started.dir)).rejects.toThrow(); + }); + + it.each(['abort', 'failure'] as const)('limits jobs to two and releases permits after %s', async (outcome) => { + const mode = path.join(dir, 'mode'); + await writeFile(mode, 'wait'); + const ffmpeg = await script('ffmpeg', 'process.exit(0);'); + const whisper = await script('whisper', `const fs = require('node:fs'); const args = process.argv.slice(2); const output = args[args.indexOf('-of') + 1]; fs.writeFileSync(${JSON.stringify(dir)} + '/' + process.pid + '.started', ''); setInterval(() => { const mode = fs.readFileSync(${JSON.stringify(mode)}, 'utf8'); if (mode === 'fail') process.exit(1); if (mode === 'ok') { fs.writeFileSync(output + '.json', JSON.stringify({transcription: []})); process.exit(0); } }, 10);`); + const local = new LocalWhisper(dir, whisper, ffmpeg); + const controller = new AbortController(); + const jobs = Promise.allSettled([0, 1].map(() => local.transcribe(new Uint8Array([1]), {signal: controller.signal}))); + try { + await expect.poll(async () => (await readdir(dir)).filter((f) => f.endsWith('.started')).length).toBe(2); + await expect(local.transcribe(new Uint8Array([1]))).rejects.toBeInstanceOf(LocalTranscriptionBusyError); + if (outcome === 'abort') controller.abort(); + else await writeFile(mode, 'fail'); + const results = await jobs; + for (const result of results) { + expect(result.status).toBe('rejected'); + if (result.status === 'rejected') expect(result.reason.message).toContain(outcome === 'abort' ? 'abort' : 'Local audio processing failed'); + } + await writeFile(mode, 'ok'); + const recovered = await Promise.all([0, 1].map(() => local.transcribe(new Uint8Array([1])))); + expect(recovered).toEqual([{text: '', segments: [], durationMs: 0}, {text: '', segments: [], durationMs: 0}]); + } finally { + await local.dispose(); + await jobs; + } + }); + + it.each([1001, 1000.6])('keeps duration integer milliseconds without a seconds round-trip (%s)', (end) => { + expect(parseWhisperOutput({transcription: [ + {offsets: {from: 0, to: end}, text: 'First'}, + {offsets: {from: 0, to: 900}, text: 'Second'}, + ]}).durationMs).toBe(1001); + }); + + it('rejects pre-aborted work, spawn failures, failed processes and malformed output', async () => { + const controller = new AbortController(); + controller.abort(); + await expect(new LocalWhisper(dir).transcribe(new Uint8Array(), {signal: controller.signal})).rejects.toMatchObject({name: 'AbortError'}); + const signal = new AbortController().signal; + await expect(runWhisperProcess(path.join(dir, 'missing'), [], signal)).rejects.toThrow(); + await expect(runWhisperProcess(process.execPath, ['-e', 'process.exit(1)'], signal)).rejects.toThrow('Local audio processing failed'); + for (const value of [null, {}, {transcription: [{text: 'bad', offsets: {from: -1, to: 100}}]}]) { + expect(() => parseWhisperOutput(value)).toThrow(); + } + expect(parseWhisperOutput({transcription: []})).toEqual({text: '', segments: [], durationMs: 0}); + }); +}); diff --git a/packages/server/src/ai/whisper.ts b/packages/server/src/ai/whisper.ts new file mode 100644 index 00000000..989b0858 --- /dev/null +++ b/packages/server/src/ai/whisper.ts @@ -0,0 +1,146 @@ +import {spawn} from 'node:child_process'; +import {constants} from 'node:fs'; +import {access, mkdtemp, readFile, rm, stat, writeFile} from 'node:fs/promises'; +import {tmpdir} from 'node:os'; +import path from 'node:path'; +import type {AiStatus, AiTranscriptionResult} from '@book.dev/sdk'; +import type {TranscriptionEngine, TranscribeOptions} from './providers'; + +export const LOCAL_TRANSCRIPTION_MAX_JOBS = 2; + +export class LocalTranscriptionBusyError extends Error { + readonly retryAfterSeconds = 5; + constructor() { + super('Local transcription is busy. Please retry shortly.'); + } +} + +export const WHISPER_MODEL = 'ggml-base.bin'; +export const WHISPER_MODEL_URL = 'https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.bin'; + +async function executable(command: string): Promise { + const candidates = path.isAbsolute(command) || command.includes(path.sep) + ? [command] : (process.env.PATH ?? '').split(path.delimiter).map((dir) => path.join(dir, command)); + for (const candidate of candidates) { + try { + await access(candidate, constants.X_OK); + if ((await stat(candidate)).isFile()) return candidate; + } catch { /* Optional system dependency. */ } + } + return null; +} + +/** One child per operation: no listening port or native Node ABI dependency. + * SIGKILL makes cancellation deterministic even inside a native inference call. + * Wait for close before removing its private scratch directory. */ +export async function runWhisperProcess(command: string, args: string[], signal: AbortSignal): Promise { + signal.throwIfAborted(); + await new Promise((resolve, reject) => { + const child = spawn(command, args, {stdio: 'ignore', shell: false}); + const abort = () => { child.kill('SIGKILL'); }; + signal.addEventListener('abort', abort, {once: true}); + if (signal.aborted) abort(); + child.once('error', (error) => { + signal.removeEventListener('abort', abort); + reject(error); + }); + child.once('close', (code) => { + signal.removeEventListener('abort', abort); + if (signal.aborted) reject(signal.reason); + else if (code !== 0) reject(new Error('Local audio processing failed. Check the recording and Settings → AI.')); + else resolve(); + }); + }); +} + +export function parseWhisperOutput(value: unknown): AiTranscriptionResult { + const data = value as {transcription?: {offsets?: {from?: number; to?: number}; text?: string}[]}; + if (!data || !Array.isArray(data.transcription)) throw new Error('Invalid whisper output'); + let durationMs = 0; + const segments = data.transcription.map((segment) => { + const start = segment?.offsets?.from; + const end = segment?.offsets?.to; + if (typeof start !== 'number' || !Number.isFinite(start) || start < 0 + || typeof end !== 'number' || !Number.isFinite(end) || end < start || typeof segment.text !== 'string') { + throw new Error('Invalid whisper segment'); + } + durationMs = Math.max(durationMs, end); + return {start: start / 1000, end: end / 1000, text: segment.text.trim()}; + }); + return {text: segments.map((s) => s.text).join(' ').trim(), segments, durationMs: Math.round(durationMs)}; +} + +/** Optional whisper.cpp + FFmpeg runtime. Models live beside chat models, but + * inference runs only on demand, releasing model memory after each recording. */ +export class LocalWhisper implements TranscriptionEngine { + private readonly active = new Set(); + private readonly pending = new Set>(); + private disposed = false; + + constructor( + private readonly modelsDir: string, + private readonly whisperCommand = process.env.OPENBOOK_WHISPER_BIN || 'whisper-cli', + private readonly ffmpegCommand = process.env.OPENBOOK_FFMPEG_BIN || 'ffmpeg', + ) {} + + async status(): Promise> { + const [whisper, ffmpeg, modelPresent] = await Promise.all([ + executable(this.whisperCommand), executable(this.ffmpegCommand), + stat(path.join(this.modelsDir, WHISPER_MODEL)).then((s) => s.isFile() && s.size > 0).catch(() => false), + ]); + const runtimeAvailable = Boolean(whisper && ffmpeg); + return { + model: WHISPER_MODEL, modelPresent, runtimeAvailable, ready: modelPresent && runtimeAvailable && !this.disposed, + downloadUrl: WHISPER_MODEL_URL, + detail: !runtimeAvailable ? 'Install whisper.cpp (whisper-cli) and FFmpeg on the server, then return to Settings → AI.' + : !modelPresent ? 'Download Whisper base in Settings → AI to enable local transcription.' : undefined, + }; + } + + async resolve(): Promise { + return (await this.status()).ready ? this : null; + } + + transcribe(bytes: Uint8Array, opts: TranscribeOptions = {}): Promise { + if (opts.signal?.aborted) return Promise.reject(opts.signal.reason); + // This set is a non-queuing semaphore shared by all requests to this server. + // Reserve synchronously, before any await or temporary-file/process creation; + // keep the permit until perform's finally has removed the scratch directory. + if (this.active.size >= LOCAL_TRANSCRIPTION_MAX_JOBS) return Promise.reject(new LocalTranscriptionBusyError()); + const controller = new AbortController(); + const signal = opts.signal ? AbortSignal.any([opts.signal, controller.signal]) : controller.signal; + this.active.add(controller); + const operation = this.perform(bytes, signal).finally(() => { + this.active.delete(controller); + this.pending.delete(operation); + }); + this.pending.add(operation); + return operation; + } + + private async perform(bytes: Uint8Array, signal: AbortSignal): Promise { + signal.throwIfAborted(); + if (this.disposed) throw new Error('Local transcription has stopped.'); + const dir = await mkdtemp(path.join(tmpdir(), 'openbook-whisper-')); + try { + const input = path.join(dir, 'input'); + const wav = path.join(dir, 'audio.wav'); + const output = path.join(dir, 'result'); + await writeFile(input, bytes, {signal}); + // Probe by content, never by an untrusted filename; accept browser WebM, + // MP4, Ogg and WAV alike. Restrict nested input protocols to local files. + await runWhisperProcess(this.ffmpegCommand, ['-nostdin', '-v', 'error', '-protocol_whitelist', 'file,pipe', '-i', input, '-vn', '-ar', '16000', '-ac', '1', '-c:a', 'pcm_s16le', wav], signal); + await runWhisperProcess(this.whisperCommand, ['-m', path.join(this.modelsDir, WHISPER_MODEL), '-f', wav, '-l', 'auto', '-oj', '-of', output], signal); + signal.throwIfAborted(); + return parseWhisperOutput(JSON.parse(await readFile(`${output}.json`, 'utf8'))); + } finally { + await rm(dir, {recursive: true, force: true}); + } + } + + async dispose(): Promise { + this.disposed = true; + for (const controller of this.active) controller.abort(); + await Promise.allSettled(this.pending); + } +} diff --git a/packages/server/src/server.ts b/packages/server/src/server.ts index bfbff2ba..a236505e 100644 --- a/packages/server/src/server.ts +++ b/packages/server/src/server.ts @@ -6,6 +6,7 @@ import {PageStore, PAGE_VERSION_KEEP, PAGE_VERSION_MAX_AGE_MS} from './store'; import {PageHub} from './hub'; import {BookMirror, MirrorLockedError, WriteBudgetError} from './mirror'; import {AiService} from './ai/service'; +import {LocalWhisper} from './ai/whisper'; import {McpClientManager} from './ai/mcpClients'; import {AiUsageLog} from './ai/usage'; import {IdentityService} from './instanceConfig'; @@ -436,7 +437,8 @@ export async function startServer(opts: StartOptions): Promise { // (server mode). The subsystem is inert until configured via /api/ai. const modelsDir = process.env.OPENBOOK_MODELS_DIR || (opts.dataDir ? path.join(opts.dataDir, 'models') : path.join(os.homedir(), '.openbook', 'models')); - const ai = new AiService(db, modelsDir); + const whisper = new LocalWhisper(modelsDir); + const ai = new AiService(db, modelsDir, () => whisper.resolve(), whisper); // External-tools (MCP client) manager (AGENT-3): owned beside AiService, pools // connections to admin-registered MCP servers and hands the agent route // namespaced `mcp__*` tools. Inert until an admin configures + enables a server diff --git a/packages/server/src/transcription.test.ts b/packages/server/src/transcription.test.ts index 6d6fce60..1f5b5a92 100644 --- a/packages/server/src/transcription.test.ts +++ b/packages/server/src/transcription.test.ts @@ -11,6 +11,9 @@ import {AiService} from './ai/service'; import {MockEngine, OpenAiCompatEngine} from './ai/providers'; import {AiUsageLog} from './ai/usage'; import {LocalDataClient} from './localClient'; +import {LocalWhisper, WHISPER_MODEL, WHISPER_MODEL_URL} from './ai/whisper'; +import {LOCAL_TRANSCRIPTION_RATE_LIMIT} from './ai/routes'; +import {readFile, readdir, writeFile} from 'node:fs/promises'; let db: PgliteDb; let store: PageStore; @@ -119,6 +122,134 @@ describe('transcription contract', () => { expect(local).toHaveBeenCalledOnce(); }); + it('uses the managed resolver fallback and exposes actionable local state through aiStatus', async () => { + const local = new LocalWhisper(dir, join(dir, 'missing'), join(dir, 'missing-ffmpeg')); + const service = new AiService(db, dir, () => local.resolve(), local); + const app = appWith(service); + expect((await post(app)).status).toBe(400); + expect((await (await post(app)).json()).error).toContain('Settings → AI'); + const status = await app.request(API.aiStatus, {headers: {...headers, [LOCAL_OWNER_HEADER]: secret}}); + expect((await status.json()).transcription).toMatchObject({modelPresent: false, runtimeAvailable: false, ready: false, downloadUrl: WHISPER_MODEL_URL}); + await service.setConfig({provider: 'mock'}); + expect((await post(app)).status).toBe(200); + await service.dispose(); + }); + + it('logs local transcription as free and downloads its model without selecting it for chat', async () => { + const service = new AiService(db, dir, async () => new MockEngine()); + const usage = new AiUsageLog(store); + expect((await post(appWith(service, usage))).status).toBe(200); + expect((await usage.report()).rows?.[0]).toMatchObject({provider: 'local', model: WHISPER_MODEL, kind: 'transcribe', cost: 0}); + await service.setConfig({provider: 'llama'}); + vi.stubGlobal('fetch', vi.fn(async () => new Response('model bytes'))); + const app = appWith(service); + const download = await app.request(API.aiModelDownload, {method: 'POST', headers: {...headers, [LOCAL_OWNER_HEADER]: secret}, body: JSON.stringify({url: WHISPER_MODEL_URL})}); + expect(download.status).toBe(200); + await expect.poll(async () => (await service.status()).download?.done).toBe(true); + expect(await readFile(join(dir, WHISPER_MODEL), 'utf8')).toBe('model bytes'); + expect((await service.getConfig()).model).toBeUndefined(); + expect(await readdir(dir)).not.toContain(`${WHISPER_MODEL}.part`); + await service.dispose(); + }); + + it('caps parallel local HTTP jobs across anonymous readers at two, returning 429 with Retry-After', async () => { + await store.updateInstanceConfig({ownerSubject: 'test#owner', guestAccess: 'read'}); + await store.setPageVisibility(pageId, 'public'); + const ffmpeg = join(dir, 'ffmpeg'); + const whisper = join(dir, 'whisper'); + const release = join(dir, 'release'); + await writeFile(ffmpeg, `#!${process.execPath}\nprocess.exit(0);`, {mode: 0o700}); + await writeFile(whisper, `#!${process.execPath}\nconst fs = require('node:fs'); const args = process.argv.slice(2); const output = args[args.indexOf('-of') + 1]; fs.writeFileSync(${JSON.stringify(dir)} + '/' + process.pid + '.started', ''); setInterval(() => { if (fs.existsSync(${JSON.stringify(release)})) { fs.writeFileSync(output + '.json', JSON.stringify({transcription: [{offsets: {from: 0, to: 1001}, text: 'Local'}]})); process.exit(0); } }, 10);`, {mode: 0o700}); + await writeFile(join(dir, WHISPER_MODEL), 'test model'); + const local = new LocalWhisper(dir, whisper, ffmpeg); + const service = new AiService(db, dir, () => local.resolve(), local); + const app = appWith(service); + const request = (ip: string) => app.request(API.aiTranscribe, { + method: 'POST', headers: {...headers, [FORWARDED_HEADER]: '1'}, body: JSON.stringify({assetId, pageId}), + }, {incoming: {socket: {remoteAddress: ip}}}); + const accepted = [request('192.0.2.1'), request('192.0.2.2')]; + try { + await expect.poll(async () => (await readdir(dir)).filter((f) => f.endsWith('.started')).length).toBe(2); + const busy = await request('192.0.2.3'); + expect(busy.status).toBe(429); + expect(busy.headers.get('Retry-After')).toBe('5'); + await writeFile(release, 'go'); + for (const response of await Promise.all(accepted)) { + expect(response.status).toBe(200); + expect(await response.json()).toMatchObject({text: 'Local', durationMs: 1001}); + } + expect((await request('192.0.2.3')).status).toBe(200); + } finally { + await service.dispose(); + await Promise.all(accepted); + } + }); + + it('rate-limits local requests per socket IP, ignores spoofed forwarding headers, and leaves mock/cloud untouched', async () => { + const engine = new MockEngine(); + const transcribe = vi.spyOn(engine, 'transcribe'); + let available = true; + const service = new AiService(db, dir, async () => available ? engine : null); + const app = appWith(service); + const request = (ip = '192.0.2.1', forwardedFor = '') => app.request(API.aiTranscribe, { + method: 'POST', headers: {...headers, [LOCAL_OWNER_HEADER]: secret, 'x-forwarded-for': forwardedFor}, body: JSON.stringify({assetId, pageId}), + }, {incoming: {socket: {remoteAddress: ip}}}); + for (let i = 0; i < LOCAL_TRANSCRIPTION_RATE_LIMIT; i++) expect((await request()).status).toBe(200); + const denied = await request('192.0.2.1', '198.51.100.1'); + expect(denied.status).toBe(429); + expect(denied.headers.get('Retry-After')).toBe('60'); + expect(transcribe).toHaveBeenCalledTimes(LOCAL_TRANSCRIPTION_RATE_LIMIT); + expect((await request('192.0.2.2')).status).toBe(200); + available = false; + await service.setConfig({provider: 'mock'}); + expect((await request()).status).toBe(200); + await service.setConfig({provider: 'off', transcription: {provider: 'openai-compat'}}); + vi.stubGlobal('fetch', vi.fn(async () => new Response(JSON.stringify({text: 'Cloud'})))); + expect((await request()).status).toBe(200); + available = true; + await service.setConfig({provider: 'off', transcription: {provider: 'local'}}); + vi.spyOn(Date, 'now').mockReturnValue(Date.now() + 60_001); + expect((await request()).status).toBe(200); + await service.dispose(); + }); + + it('returns 400 for malformed percent-encoding in model download filenames without fetching', async () => { + const fetchSpy = vi.spyOn(globalThis, 'fetch'); + const app = appWith(); + for (const filename of ['bad%.bin', 'bad%ZZ.bin', 'bad%E0%A4%A.bin']) { + const response = await app.request(API.aiModelDownload, { + method: 'POST', headers: {...headers, [LOCAL_OWNER_HEADER]: secret}, + body: JSON.stringify({url: `https://example.test/${filename}`}), + }); + expect(response.status).toBe(400); + expect(await response.json()).toEqual({error: 'Invalid model download URL or filename encoding.'}); + } + expect(fetchSpy).not.toHaveBeenCalled(); + expect((await ai.status()).download).toBeUndefined(); + }); + + // Opt-in: OPENBOOK_TEST_WHISPER=1, OPENBOOK_MODELS_DIR containing ggml-base.bin, + // whisper-cli + ffmpeg on PATH (or OPENBOOK_WHISPER_BIN / OPENBOOK_FFMPEG_BIN). + it.skipIf(process.env.OPENBOOK_TEST_WHISPER !== '1')('native whisper: POST transcribes synthesized WAV with no cloud keys', async () => { + const local = new LocalWhisper(process.env.OPENBOOK_MODELS_DIR || join(dir, 'models')); + expect((await local.status()).ready).toBe(true); + const wav = Buffer.alloc(44 + 16000 * 2); + wav.write('RIFF'); wav.writeUInt32LE(wav.length - 8, 4); wav.write('WAVEfmt ', 8); + wav.writeUInt32LE(16, 16); wav.writeUInt16LE(1, 20); wav.writeUInt16LE(1, 22); + wav.writeUInt32LE(16000, 24); wav.writeUInt32LE(32000, 28); wav.writeUInt16LE(2, 32); wav.writeUInt16LE(16, 34); + wav.write('data', 36); wav.writeUInt32LE(wav.length - 44, 40); + assetId = (await store.putAsset(wav, 'audio/wav')).id; + await store.refAsset(assetId, pageId); + const service = new AiService(db, dir, () => local.resolve(), local); + try { + const response = await post(appWith(service)); + expect(response.status).toBe(200); + expect(await response.json()).toMatchObject({text: expect.any(String), segments: expect.any(Array), durationMs: expect.any(Number)}); + } finally { + await service.dispose(); + } + }, 120_000); + it('returns identical 404s for missing, unreadable, unreferenced, and unrelated assets/pages', async () => { await ai.setConfig({provider: 'mock'}); await store.updateInstanceConfig({ownerSubject: 'test#owner', guestAccess: 'read'}); diff --git a/packages/ui/src/components/AiSettings.tsx b/packages/ui/src/components/AiSettings.tsx index d4b868c1..88f3e421 100644 --- a/packages/ui/src/components/AiSettings.tsx +++ b/packages/ui/src/components/AiSettings.tsx @@ -27,7 +27,7 @@ function normalize(c: AiConfig): AiConfig { // the `apiKeySet` signal so the form knows a key is stored without holding it. providers[c.provider] = {model: c.model, baseUrl: c.baseUrl, apiKeySet: c.apiKeySet, autoStart: c.autoStart}; } - return {provider: c.provider, providers, effort: c.effort, thinking: c.thinking}; + return {provider: c.provider, providers, effort: c.effort, thinking: c.thinking, transcription: c.transcription}; } /** @@ -273,6 +273,15 @@ export default function AiSettings() { return ( + {status?.transcription && ( + +

{t(status.transcription.modelPresent ? 'ai.transcription.modelPresent' : 'ai.transcription.modelAbsent')} {t(status.transcription.ready ? 'ai.transcription.ready' : status.transcription.runtimeAvailable ? 'ai.transcription.modelMissing' : 'ai.transcription.runtimeMissing')}

+ + {download?.url === status.transcription.downloadUrl && download.error &&

{download.error}

} +
+ )}
{providers.map((p) => ( diff --git a/packages/ui/src/i18n/__tests__/i18n.test.ts b/packages/ui/src/i18n/__tests__/i18n.test.ts index 49e49b59..5386fac3 100644 --- a/packages/ui/src/i18n/__tests__/i18n.test.ts +++ b/packages/ui/src/i18n/__tests__/i18n.test.ts @@ -18,6 +18,17 @@ describe('t', () => { expect(t('common.cancel')).toBe('取消'); }); + it('ships local transcription copy and progress in all four locales', () => { + for (const catalog of [en, de, ja, zh]) { + expect(Object.keys(catalog.ai?.transcription ?? {}).sort()).toEqual(Object.keys(en.ai.transcription).sort()); + expect(Object.values(catalog.ai?.transcription ?? {}).every((value) => typeof value === 'string' && value.length > 0)).toBe(true); + } + for (const locale of ['en', 'de', 'ja', 'zh'] as const) { + setLocale(locale); + expect(t('ai.transcription.downloadingProgress', {progress: 42})).toContain('42%'); + } + }); + it('interpolates {var} placeholders', () => { setLocale('en'); expect(t('mention.create', {name: 'Roadmap'})).toBe('Create subpage “Roadmap”'); diff --git a/packages/ui/src/i18n/messages/de.ts b/packages/ui/src/i18n/messages/de.ts index 74d88ec2..c4480449 100644 --- a/packages/ui/src/i18n/messages/de.ts +++ b/packages/ui/src/i18n/messages/de.ts @@ -532,6 +532,18 @@ export const de: PartialMessages = { localNetwork: 'Lokales Netzwerk', }, ai: { + transcription: { + title: 'Lokale Transkription', + description: 'Aufnahmen werden standardmäßig lokal mit Whisper transkribiert, ohne Cloud-Schlüssel. Whisper base ist mehrsprachig (~142 MiB).', + modelPresent: 'Modell heruntergeladen.', + modelAbsent: 'Modell nicht heruntergeladen.', + ready: 'Bereit zur Transkription.', + runtimeMissing: 'Installiere whisper-cli und FFmpeg auf dem Server, um die lokale Transkription zu aktivieren.', + modelMissing: 'Lade unten das Modell herunter, um die lokale Transkription zu aktivieren.', + download: 'Whisper base herunterladen', + downloading: 'Wird heruntergeladen…', + downloadingProgress: 'Download: {progress}%', + }, title: 'KI', description: 'Ein optionales lokales Modell ermöglicht Notizsuche, Aufgabenplanung und Textvervollständigung. Alles läuft auf deinem Gerät.', providerLabel: 'Engine', diff --git a/packages/ui/src/i18n/messages/en.ts b/packages/ui/src/i18n/messages/en.ts index 298154c9..d7375821 100644 --- a/packages/ui/src/i18n/messages/en.ts +++ b/packages/ui/src/i18n/messages/en.ts @@ -846,6 +846,18 @@ export const en = { removeBody: 'This only removes it from this device’s library list. The server and its data are untouched.', }, ai: { + transcription: { + title: 'Local transcription', + description: 'Recordings use Whisper locally by default, with no cloud key. Whisper base is multilingual (~142 MiB).', + modelPresent: 'Model downloaded.', + modelAbsent: 'Model not downloaded.', + ready: 'Ready to transcribe.', + runtimeMissing: 'Install whisper-cli and FFmpeg on the server to enable local transcription.', + modelMissing: 'Download the model below to enable local transcription.', + download: 'Download Whisper base', + downloading: 'Downloading…', + downloadingProgress: 'Downloading {progress}%', + }, title: 'AI', description: 'An optional model powers note search, task breakdown, and writing help. Run it locally — nothing leaves your machine — or connect the Claude API.', providerLabel: 'Engine', diff --git a/packages/ui/src/i18n/messages/ja.ts b/packages/ui/src/i18n/messages/ja.ts index ab4f9a78..89e748fa 100644 --- a/packages/ui/src/i18n/messages/ja.ts +++ b/packages/ui/src/i18n/messages/ja.ts @@ -527,6 +527,18 @@ export const ja: PartialMessages = { localNetwork: 'ローカルネットワーク', }, ai: { + transcription: { + title: 'ローカル文字起こし', + description: '録音は標準でWhisperを使ってローカルで文字起こしされます。クラウドのAPIキーは不要です。Whisper baseは多言語対応です(約142 MiB)。', + modelPresent: 'モデルをダウンロード済みです。', + modelAbsent: 'モデルが未ダウンロードです。', + ready: '文字起こしの準備ができています。', + runtimeMissing: 'ローカル文字起こしを有効にするには、サーバーにwhisper-cliとFFmpegをインストールしてください。', + modelMissing: 'ローカル文字起こしを有効にするには、以下のモデルをダウンロードしてください。', + download: 'Whisper baseをダウンロード', + downloading: 'ダウンロード中…', + downloadingProgress: 'ダウンロード中 {progress}%', + }, title: 'AI', description: 'オプションのローカルモデルで、ノート検索・タスク分解・文書補完ができます。すべて端末内で動作します。', providerLabel: 'エンジン', diff --git a/packages/ui/src/i18n/messages/zh.ts b/packages/ui/src/i18n/messages/zh.ts index 604f6096..769bed00 100644 --- a/packages/ui/src/i18n/messages/zh.ts +++ b/packages/ui/src/i18n/messages/zh.ts @@ -526,6 +526,18 @@ export const zh: PartialMessages = { localNetwork: '局域网', }, ai: { + transcription: { + title: '本地转录', + description: '录音默认使用 Whisper 在本地转录,无需云端密钥。Whisper base 支持多种语言(约142 MiB)。', + modelPresent: '模型已下载。', + modelAbsent: '模型尚未下载。', + ready: '已准备好转录。', + runtimeMissing: '请在服务器上安装 whisper-cli 和 FFmpeg 以启用本地转录。', + modelMissing: '请下载下方模型以启用本地转录。', + download: '下载 Whisper base', + downloading: '正在下载…', + downloadingProgress: '正在下载 {progress}%', + }, title: 'AI', description: '可选的本地模型支持笔记搜索、任务分解和文档补全。一切都在你的设备上运行。', providerLabel: '引擎',