From 69f01e0074451e11e58c370184649c24c1a6dbd2 Mon Sep 17 00:00:00 2001 From: Eliot Lim Date: Sat, 3 Oct 2026 21:47:24 +0800 Subject: [PATCH 1/5] feat(server,sdk): add asset transcription service and backend (MEET-2) --- _brief.md | 29 ++++ packages/sdk/src/ai.ts | 19 +++ packages/sdk/src/client.ts | 6 + packages/sdk/src/index.ts | 2 + packages/sdk/src/routes.ts | 1 + packages/server/src/ai/providers.ts | 50 +++++- packages/server/src/ai/routes.ts | 54 ++++++- packages/server/src/ai/service.ts | 29 +++- packages/server/src/ai/usage.ts | 6 +- packages/server/src/localClient.ts | 5 + packages/server/src/transcription.test.ts | 187 ++++++++++++++++++++++ 11 files changed, 382 insertions(+), 6 deletions(-) create mode 100644 _brief.md create mode 100644 packages/server/src/transcription.test.ts diff --git a/_brief.md b/_brief.md new file mode 100644 index 00000000..eac54b32 --- /dev/null +++ b/_brief.md @@ -0,0 +1,29 @@ +# MEET-2 — Transcription service + OpenAI-compatible backend + +You are a Worker agent on the OpenBook team. Work ONLY in this worktree (`/Users/eliot/Workspaces/OpenBook-wt-meet-2`, branch `feat/meet-2-transcribe`). Do NOT push. Do NOT touch main. Conventional commits (`feat(server,sdk): … (MEET-2)`), committed incrementally. + +## Task +Add a transcription capability to the AI layer. No engine has audio today. Key anchors (recon-time line hints — trust the code): +- Engines: `packages/server/src/ai/providers.ts` — `MockEngine` :108, `OpenAiCompatEngine` :196, `AnthropicEngine` :620, factory `createEngine` :821. +- Config: `AiConfig` `packages/sdk/src/ai.ts:66`, persisted in DB `settings` key `'ai'` (`ai/service.ts:115-123`); API keys are write-only over the wire (`resolveKey` service.ts:31-37 — blank keeps, null clears). Preserve those semantics for any new key field. +- Routes: `packages/server/src/ai/routes.ts` — `aiComplete` :137 is the request-scoped shape to mirror; paid-provider gate `requirePaidInferenceAccess` :412; usage logging `ai/usage.ts`. +- Access: default-deny via `access.ts` (`requireAccess`); unreadable → 404. + +## Design (contract for MEET-3 local whisper and MEET-5 UI — keep the interface clean) +1. Extend `AiConfig` with a `transcription` section: `{ provider: 'off' | 'local' | 'openai-compat', baseUrl?, model?, apiKey? }`. **Resolution order: explicit cloud config > local (arrives in MEET-3; stub the enum/dispatch now) > clear actionable error pointing at Settings → AI.** Local is the product default — cloud only when explicitly configured. +2. New `transcribe` capability speaking the OpenAI-compatible `POST /v1/audio/transcriptions` multipart shape (works for OpenAI cloud and local servers like faster-whisper/speaches — ONE code path). Request `verbose_json`/segments where available. +3. Server route (under ai/routes.ts): POST taking `{ assetId, pageId }` (bytes already in the asset store — fetch via the store with the caller's read access enforced), returning `{ text, segments?: [{start, end, text}], durationMs }`. Request-scoped like aiComplete. Do NOT build a job queue. +4. Cloud transcription goes through `requirePaidInferenceAccess`; `local` will be ungated. Log usage rows. +5. sdk: `transcribeAsset(assetId, pageId)` on the HTTP client (pattern near `client.ts:1571`). `LocalDataClient` (`server/localClient.ts:888-932`) keeps rejecting AI calls — extend its rejection list to the new method. + +## Acceptance (each maps to a test) +- MockEngine grows a deterministic `transcribe` for tests. +- Unit tests: config round-trip incl. write-only key semantics; route happy path via mock; paid-gate enforced for cloud provider; 404 on missing/unreadable asset; `off`/unconfigured → actionable 4xx error. +- NOTE: audio MIME allowlist work is MEET-1 (parallel branch) — do not depend on it; transcribe takes the asset bytes regardless of stored MIME. + +## Definition of done +- `pnpm verify` green FOREGROUND in this worktree, output in report. (Provisioned; artifact-failure symptom → rebuild viewer/server/mcp bundles.) +- All committed, not pushed. Write `_report.md` (outcome first, head sha, criterion→test map, interface summary for MEET-3/MEET-5 consumers, deviations, open questions) and reply with it. Terse. + +## Rules +Never poll external state. Stuck after a real attempt → commit + report where wedged. Never delete/weaken existing tests (reviewers check the commit RANGE). diff --git a/packages/sdk/src/ai.ts b/packages/sdk/src/ai.ts index ff0fcf70..5212ad79 100644 --- a/packages/sdk/src/ai.ts +++ b/packages/sdk/src/ai.ts @@ -63,7 +63,26 @@ export interface AiProviderSettings { autoStart?: boolean; } +/** Audio configuration is independent of the chat engine. Omitted means local. + * Keys follow AiProviderSettings.apiKey preserve/set/clear semantics. */ +export interface AiTranscriptionConfig { + provider: 'off' | 'local' | 'openai-compat'; + baseUrl?: string; + model?: string; + apiKey?: string | null; + apiKeySet?: boolean; +} + +export interface AiTranscriptionResult { + text: string; + /** Segment offsets are seconds from the start of the recording. */ + segments?: Array<{start: number; end: number; text: string}>; + /** Audio duration in milliseconds (0 if the backend does not report it). */ + durationMs: number; +} + export interface AiConfig { + transcription?: AiTranscriptionConfig; /** The default provider — used unless an agent run overrides it. */ provider: AiProvider; /** Per-provider settings, so every provider can be configured at once. */ diff --git a/packages/sdk/src/client.ts b/packages/sdk/src/client.ts index c1657696..013fb178 100644 --- a/packages/sdk/src/client.ts +++ b/packages/sdk/src/client.ts @@ -5,6 +5,7 @@ import type { AgentChatMessage, AgentChatOptions, AiConfig, + AiTranscriptionResult, AiPricingResponse, AiPricingTable, AiUsageResponse, @@ -208,6 +209,7 @@ export interface DataClient { aiSearch(query: string, limit?: number): Promise; aiTasks(goal: string, context?: string): Promise; aiDownloadModel(url?: string): Promise; + transcribeAsset(assetId: string, pageId: string): Promise; aiComplete(text: string, onToken: (token: string) => void, opts?: {instruction?: string; signal?: AbortSignal}): Promise; aiGenerate(prompt: string, onToken: (token: string) => void, opts?: {system?: string; maxTokens?: number; signal?: AbortSignal}): Promise; /** @@ -2146,6 +2148,10 @@ export class HttpDataClient implements DataClient { return this.request('POST', API.aiTasks, {goal, context}); } + async transcribeAsset(assetId: string, pageId: string): Promise { + return this.request('POST', API.aiTranscribe, {assetId, pageId}); + } + async aiDownloadModel(url?: string): Promise { return this.request('POST', API.aiModelDownload, {url}); } diff --git a/packages/sdk/src/index.ts b/packages/sdk/src/index.ts index 6ca60969..53fe6687 100644 --- a/packages/sdk/src/index.ts +++ b/packages/sdk/src/index.ts @@ -690,6 +690,8 @@ export type { InterviewStep, AiProvider, AiConfig, + AiTranscriptionConfig, + AiTranscriptionResult, AiProviderSettings, AiEffort, AiSkill, diff --git a/packages/sdk/src/routes.ts b/packages/sdk/src/routes.ts index 6474c800..33df46d8 100644 --- a/packages/sdk/src/routes.ts +++ b/packages/sdk/src/routes.ts @@ -137,6 +137,7 @@ export const API = { aiTasks: '/api/ai/tasks', /** Continue/complete document text: `POST` `{text, instruction?}` → SSE. */ aiComplete: '/api/ai/complete', + aiTranscribe: '/api/ai/transcribe', /** Download a model file for the in-process engine: `POST` `{url?}`. */ aiModelDownload: '/api/ai/models/download', /** The agent harness: `POST` `{messages, effort?, thinking?, skills?}` → SSE tool/reasoning/proposal/final events. */ diff --git a/packages/server/src/ai/providers.ts b/packages/server/src/ai/providers.ts index 74727e3b..5155fa52 100644 --- a/packages/server/src/ai/providers.ts +++ b/packages/server/src/ai/providers.ts @@ -1,7 +1,7 @@ import {spawn, type ChildProcess} from 'node:child_process'; import {existsSync} from 'node:fs'; import path from 'node:path'; -import {providerSettings, type AiConfig, type AiProvider} from '@book.dev/sdk'; +import {providerSettings, type AiConfig, type AiProvider, type AiTranscriptionResult} from '@book.dev/sdk'; /** * Inference engines behind one interface. Generation streams tokens; @@ -84,7 +84,19 @@ export interface GenerateOptions { signal?: AbortSignal; } +export interface TranscribeOptions { + filename?: string; + mime?: string; + signal?: AbortSignal; +} + +/** MEET-3 implements this capability; no chat/embedding engine is required. */ +export interface TranscriptionEngine { + transcribe(bytes: Uint8Array, opts?: TranscribeOptions): Promise; +} + export interface AiEngine { + transcribe?: TranscriptionEngine['transcribe']; readonly kind: string; /** Throws (with a user-readable message) when the engine can't run. */ ensureReady(): Promise; @@ -108,6 +120,12 @@ export interface AiEngine { export class MockEngine implements AiEngine { readonly kind = 'mock'; + async transcribe(bytes: Uint8Array, opts: TranscribeOptions = {}): Promise { + opts.signal?.throwIfAborted(); + const text = `Mock transcription (${bytes.byteLength} bytes).`; + return {text, segments: [{start: 0, end: 1, text}], durationMs: 1000}; + } + async ensureReady(): Promise { // always ready } @@ -199,10 +217,40 @@ export class OpenAiCompatEngine implements AiEngine { private readonly baseUrl: string, private readonly model: string, kind = 'openai', + private readonly apiKey?: string | null, ) { this.kind = kind; } + async transcribe(bytes: Uint8Array, opts: TranscribeOptions = {}): Promise { + const form = new FormData(); + const extensions: Record = {'audio/webm': 'webm', 'video/webm': 'webm', 'audio/wav': 'wav', 'audio/x-wav': 'wav', 'audio/mpeg': 'mp3', 'audio/mp4': 'm4a', 'audio/ogg': 'ogg', 'audio/flac': 'flac'}; + const filename = opts.filename || `audio.${extensions[opts.mime?.split(';')[0] ?? ''] ?? 'bin'}`; + form.append('file', new Blob([new Uint8Array(bytes)], {type: opts.mime || 'application/octet-stream'}), filename); + form.append('model', this.model || 'whisper-1'); + form.append('response_format', 'verbose_json'); + form.append('timestamp_granularities[]', 'segment'); + const base = this.baseUrl.replace(/\/+$/, '').replace(/\/v1$/, ''); + const res = await fetch(`${base}/v1/audio/transcriptions`, { + method: 'POST', + headers: this.apiKey ? {Authorization: `Bearer ${this.apiKey}`} : {}, + body: form, + signal: opts.signal ? AbortSignal.any([opts.signal, AbortSignal.timeout(300_000)]) : AbortSignal.timeout(300_000), + }); + // Do not echo upstream bodies: they can contain credentials or recording text. + if (!res.ok) throw new Error(`Transcription provider returned HTTP ${res.status}`); + const data = await res.json() as {text?: unknown; duration?: unknown; segments?: unknown}; + if (typeof data.text !== 'string') throw new Error('Invalid transcription provider response'); + const segments: AiTranscriptionResult['segments'] = Array.isArray(data.segments) + ? data.segments.filter((s): s is {start: number; end: number; text: string} => + s && Number.isFinite(s.start) && s.start >= 0 && Number.isFinite(s.end) && s.end >= s.start && typeof s.text === 'string') + .map(({start, end, text}) => ({start, end, text})) + : undefined; + const duration = typeof data.duration === 'number' && Number.isFinite(data.duration) && data.duration >= 0 + ? data.duration : segments?.reduce((end, s) => Math.max(end, s.end), 0) ?? 0; + return {text: data.text, ...(segments ? {segments} : {}), durationMs: Math.round(duration * 1000)}; + } + async ensureReady(): Promise { const res = await fetch(`${this.baseUrl}/v1/models`, {signal: AbortSignal.timeout(3000)}).catch(() => null); if (!res?.ok) { diff --git a/packages/server/src/ai/routes.ts b/packages/server/src/ai/routes.ts index 2b09d54a..3cb003ef 100644 --- a/packages/server/src/ai/routes.ts +++ b/packages/server/src/ai/routes.ts @@ -6,7 +6,7 @@ import type {PageStore} from '../store'; import type {AppEnv} from '../appEnv'; import {isLocalInstanceOwner, requireAuthenticatedRead, requireCreate, requireInstanceAdmin, requireInstanceOwner} from '../access'; import {AgentRunner, type AgentMessage} from './agent'; -import type {AiService} from './service'; +import {TranscriptionConfigError, type AiService} from './service'; import {McpConfigError, type ExternalAgentTool, type McpClientManager} from './mcpClients'; import type {TokenUsage} from './providers'; import type {AiUsageLog, UsageKind} from './usage'; @@ -61,6 +61,22 @@ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore if (!['off', 'mock', 'llama', 'mlx', 'openai', 'claude'].includes(body.provider)) { return c.json({error: `Unknown provider: ${String(body.provider)}`}, 400); } + if (body.transcription !== undefined) { + const audio = body.transcription; + if (!audio || !['off', 'local', 'openai-compat'].includes(audio.provider) + || [audio.baseUrl, audio.model].some((v) => v !== undefined && typeof v !== 'string') + || (audio.apiKey !== undefined && audio.apiKey !== null && typeof audio.apiKey !== 'string')) { + return c.json({error: 'Invalid transcription configuration'}, 400); + } + if (audio.baseUrl) { + try { + const url = new URL(audio.baseUrl); + if (!['http:', 'https:'].includes(url.protocol) || url.username || url.password) throw new Error(); + } catch { + return c.json({error: 'Transcription baseUrl must be an HTTP(S) URL without embedded credentials'}, 400); + } + } + } // Redact the echoed config too: a blank-on-save PRESERVES the stored key // (see `AiService.setConfig`), so returning the saved config raw would hand a // previously-stored key back to the writer that just blanked the field. The key @@ -160,6 +176,36 @@ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore }); }); + app.post(API.aiTranscribe, async (c) => { + const body = await c.req.json().catch(() => null) as {assetId?: unknown; pageId?: unknown} | null; + if (typeof body?.assetId !== 'string' || typeof body.pageId !== 'string' || !body.assetId || !body.pageId) { + return c.json({error: 'assetId and pageId are required'}, 400); + } + const {assetId, pageId} = body; + if (!/^[0-9a-f]{64}$/.test(assetId) || !/^[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}$/i.test(pageId)) { + return c.json({error: 'asset not found'}, 404); + } + const principal = c.get('principal'); + if (!await store.getPageFor(principal, pageId) + || !(await store.pagesReferencingAsset(assetId)).includes(pageId)) { + return c.json({error: 'asset not found'}, 404); + } + const asset = await store.getAssetFor(principal, assetId); + if (!asset) return c.json({error: 'asset not found'}, 404); + try { + const {engine, provider, model} = await ai.transcriptionBackend(); + await requirePaidInferenceAccess(c, store, provider === 'openai-compat' ? 'openai' : 'off'); + const result = await engine.transcribe(asset.bytes, {mime: asset.mime, signal: c.req.raw.signal}); + // Audio backends do not report token counts. Keep unknown cloud cost null. + await aiUsage?.log({provider, model, kind: 'transcribe', principal, usage: {inputTokens: 0, outputTokens: 0}}); + return c.json(result); + } catch (err) { + if (err instanceof HTTPException) throw err; + if (err instanceof TranscriptionConfigError) return c.json({error: err.message}, 400); + return c.json({error: 'Transcription failed. Check the provider in Settings → AI and retry.'}, 502); + } + }); + app.post(API.aiModelDownload, async (c) => { // Fetches a caller-supplied URL onto the server's disk (SSRF + disk-fill // surface) — only the trusted instance owner may supply it. @@ -378,6 +424,12 @@ function redactAiConfig(config: AiConfig): AiConfig { delete redacted.apiKey; if (hasKey(config.apiKey)) redacted.apiKeySet = true; else delete redacted.apiKeySet; + if (config.transcription) { + redacted.transcription = {...config.transcription}; + delete redacted.transcription.apiKey; + if (hasKey(config.transcription.apiKey)) redacted.transcription.apiKeySet = true; + else delete redacted.transcription.apiKeySet; + } if (config.providers) { redacted.providers = Object.fromEntries( Object.entries(config.providers).map(([p, settings]) => { diff --git a/packages/server/src/ai/service.ts b/packages/server/src/ai/service.ts index 73351533..f03187b1 100644 --- a/packages/server/src/ai/service.ts +++ b/packages/server/src/ai/service.ts @@ -3,7 +3,7 @@ import {rename, unlink} from 'node:fs/promises'; import path from 'node:path'; import {providerSettings, type AiConfig, type AiProvider, type AiProviderSettings, type AiSearchResponse, type AiStatus, type AiTasksResponse} from '@book.dev/sdk'; import type {Db} from '../db'; -import {createEngine, type AiEngine, type GenerateOptions} from './providers'; +import {createEngine, MockEngine, OpenAiCompatEngine, type TranscriptionEngine, type AiEngine, type GenerateOptions} from './providers'; import {assembleSearchResults, bm25Scores, buildIndex, cosine, pageRowsToDocs, parseTaskList, type Bm25Index} from './search'; import {SkillStore} from './skills'; @@ -73,6 +73,13 @@ function mergeStoredKeys(prev: AiConfig, next: AiConfig): AiConfig { } merged.providers = out as AiConfig['providers']; } + if (next.transcription || prev.transcription) { + merged.transcription = {...(next.transcription ?? prev.transcription!)}; + delete merged.transcription.apiKeySet; + const key = resolveKey(prev.transcription?.apiKey, next.transcription?.apiKey); + if (key === undefined) delete merged.transcription.apiKey; + else merged.transcription.apiKey = key; + } return merged; } @@ -88,6 +95,8 @@ interface DownloadState { error?: string; } +export class TranscriptionConfigError extends Error {} + export class AiService { private config: AiConfig = DEFAULT_CONFIG; private engine: AiEngine | null = null; @@ -103,6 +112,8 @@ export class AiService { constructor( private readonly db: Db, private readonly modelsDir: string, + /** MEET-3: lazily resolve the managed local audio backend. */ + private readonly localTranscription?: () => Promise, ) { this.skills = new SkillStore(db); } @@ -140,6 +151,22 @@ export class AiService { return this.config; } + /** Resolve once per request so config changes cannot bypass the paid gate. */ + async transcriptionBackend(): Promise<{engine: TranscriptionEngine; provider: 'local' | 'mock' | 'openai-compat'; model: string}> { + const config = await this.getConfig(); + const audio = config.transcription; + if (audio?.provider === 'off') { + throw new TranscriptionConfigError('Transcription is off. Enable it in Settings → AI.'); + } + if (audio?.provider === 'openai-compat') { + return {engine: new OpenAiCompatEngine(audio.baseUrl?.trim() || 'https://api.openai.com', audio.model?.trim() || 'whisper-1', 'openai', audio.apiKey), provider: 'openai-compat', model: audio.model?.trim() || 'whisper-1'}; + } + const local = await this.localTranscription?.(); + if (local) return {engine: local, provider: 'local', model: audio?.model ?? 'local'}; + if (config.provider === 'mock') return {engine: new MockEngine(), provider: 'mock', model: 'mock'}; + throw new TranscriptionConfigError('Local transcription is unavailable. Configure transcription in Settings → AI.'); + } + async status(): Promise { await this.loadConfig(); let ready = false; diff --git a/packages/server/src/ai/usage.ts b/packages/server/src/ai/usage.ts index 80c5f1a4..289dbb35 100644 --- a/packages/server/src/ai/usage.ts +++ b/packages/server/src/ai/usage.ts @@ -68,11 +68,11 @@ const PROP = { } as const; /** The kinds of model call we attribute. */ -export type UsageKind = 'agent' | 'complete' | 'generate'; +export type UsageKind = 'agent' | 'complete' | 'generate' | 'transcribe'; /** One model call to attribute: what ran, how many tokens, and for whom. */ export interface UsageEvent { - provider: AiProvider; + provider: AiProvider | 'local' | 'openai-compat'; model: string; kind: UsageKind; usage: TokenUsage; @@ -364,7 +364,7 @@ export class AiUsageLog { // A claude call with no configured model runs on the engine's default — price // (and log) against that so cost isn't spuriously null. const model = event.model || (event.provider === 'claude' ? AnthropicEngine.DEFAULT_MODEL : ''); - const cost = this.computeCost(event.provider, model, event.usage, effective); + const cost = event.provider === 'openai-compat' ? null : event.provider === 'local' ? 0 : this.computeCost(event.provider, model, event.usage, effective); const properties: Record = { [PROP.time]: new Date().toISOString(), [PROP.user]: formatUser(event.principal), diff --git a/packages/server/src/localClient.ts b/packages/server/src/localClient.ts index d8ca4560..69001f19 100644 --- a/packages/server/src/localClient.ts +++ b/packages/server/src/localClient.ts @@ -5,6 +5,7 @@ import type { AgentTokenList, CreatedAgentToken, AiConfig, + AiTranscriptionResult, AiPricingResponse, AiPricingTable, AiUsageResponse, @@ -920,6 +921,10 @@ export class LocalDataClient implements DataClient { return Promise.reject(this.aiUnavailable()); } + transcribeAsset(): Promise { + return Promise.reject(this.aiUnavailable()); + } + aiComplete(): Promise { return Promise.reject(this.aiUnavailable()); } diff --git a/packages/server/src/transcription.test.ts b/packages/server/src/transcription.test.ts new file mode 100644 index 00000000..6d6fce60 --- /dev/null +++ b/packages/server/src/transcription.test.ts @@ -0,0 +1,187 @@ +import {mkdtempSync, rmSync} from 'node:fs'; +import {tmpdir} from 'node:os'; +import {join} from 'node:path'; +import {afterEach, beforeEach, describe, expect, it, vi} from 'vitest'; +import {API, FORWARDED_HEADER, LOCAL_OWNER_HEADER, HttpDataClient} from '@book.dev/sdk'; +import {PgliteDb} from './db'; +import {PageStore} from './store'; +import {PageHub} from './hub'; +import {createApp} from './app'; +import {AiService} from './ai/service'; +import {MockEngine, OpenAiCompatEngine} from './ai/providers'; +import {AiUsageLog} from './ai/usage'; +import {LocalDataClient} from './localClient'; + +let db: PgliteDb; +let store: PageStore; +let ai: AiService; +let dir: string; +let pageId: string; +let assetId: string; +const secret = 'transcription-owner'; +const headers = {'content-type': 'application/json', 'X-OpenBook-Client': '1'}; +const snapshot = () => ({editorjs: {blocks: []}, values: [], names: []}); + +beforeEach(async () => { + dir = mkdtempSync(join(tmpdir(), 'meet2-')); + db = await PgliteDb.create(dir); + store = new PageStore(db); + await store.migrate(); + ai = new AiService(db, join(dir, 'models')); + pageId = (await store.upsertPage({name: 'Recording', data: snapshot()})).id; + assetId = (await store.putAsset(new Uint8Array([1, 2, 3]), 'application/octet-stream')).id; + await store.refAsset(assetId, pageId); +}); +afterEach(async () => { + vi.restoreAllMocks(); + vi.unstubAllGlobals(); + await store.close(); + rmSync(dir, {recursive: true, force: true}); +}); +const appWith = (service = ai, usage?: AiUsageLog) => createApp(store, service, new PageHub(), {localOwnerSecret: secret, aiUsage: usage}); +const post = (app = appWith(), body: unknown = {assetId, pageId}, guest = false) => app.request(API.aiTranscribe, { + method: 'POST', headers: {...headers, ...(guest ? {[FORWARDED_HEADER]: '1'} : {[LOCAL_OWNER_HEADER]: secret})}, body: JSON.stringify(body), +}); + +describe('transcription contract', () => { + it('round-trips and redacts keys, preserves blank/omitted keys, replaces and clears explicitly', async () => { + const app = appWith(); + const save = async (transcription?: unknown) => { + const res = await app.request(API.aiConfig, {method: 'PUT', headers: {...headers, [LOCAL_OWNER_HEADER]: secret}, body: JSON.stringify({provider: 'off', transcription})}); + expect(res.status).toBe(200); + const value = await res.json(); + expect(JSON.stringify(value)).not.toContain('secret-key'); + expect(value.transcription.apiKey).toBeUndefined(); + return value; + }; + expect((await save({provider: 'openai-compat', baseUrl: 'https://example.test/v1', model: 'whisper', apiKey: ' secret-key ', apiKeySet: true})).transcription.apiKeySet).toBe(true); + expect((await new AiService(db, dir).getConfig()).transcription).toEqual({provider: 'openai-compat', baseUrl: 'https://example.test/v1', model: 'whisper', apiKey: 'secret-key'}); + for (const apiKey of ['', ' ', undefined]) { + await save({provider: 'openai-compat', apiKey}); + expect((await ai.getConfig()).transcription?.apiKey).toBe('secret-key'); + } + await save(); + expect((await ai.getConfig()).transcription?.apiKey).toBe('secret-key'); + await save({provider: 'local', apiKey: 'replacement-secret-key'}); + expect((await ai.getConfig()).transcription?.apiKey).toBe('replacement-secret-key'); + const status = await app.request(API.aiStatus, {headers: {...headers, [LOCAL_OWNER_HEADER]: secret}}); + expect((await status.json()).config.transcription).toEqual({provider: 'local', apiKeySet: true}); + expect((await save({provider: 'local', apiKey: null, apiKeySet: true})).transcription.apiKeySet).toBeUndefined(); + expect((await new AiService(db, dir).getConfig()).transcription?.apiKey).toBeUndefined(); + }); + + it('returns the deterministic mock result and logs one attributed usage row', async () => { + await ai.setConfig({provider: 'mock'}); + const usage = new AiUsageLog(store); + const res = await post(appWith(ai, usage)); + expect(res.status).toBe(200); + expect(await res.json()).toEqual(await new MockEngine().transcribe(new Uint8Array([1, 2, 3]))); + const report = await usage.report(); + expect(report.rows).toHaveLength(1); + expect(report.rows?.[0]).toMatchObject({provider: 'mock', kind: 'transcribe', inputTokens: 0, outputTokens: 0, cost: 0}); + }); + + it('denies cloud to claimed-instance guests before sending bytes, even when chat is mock', async () => { + await store.updateInstanceConfig({ownerSubject: 'test#owner', guestAccess: 'write'}); + await store.setPageVisibility(pageId, 'public'); + await ai.setConfig({provider: 'mock', transcription: {provider: 'openai-compat'}}); + const fetchSpy = vi.spyOn(globalThis, 'fetch'); + expect((await post(appWith(), undefined, true)).status).toBe(403); + expect(fetchSpy).not.toHaveBeenCalled(); + }); + + it('allows authenticated cloud transcription, attributes unknown cost, and sanitizes failures', async () => { + await ai.setConfig({provider: 'off', transcription: {provider: 'openai-compat', model: 'whisper-1', apiKey: 'secret'}}); + const usage = new AiUsageLog(store); + const fetchMock = vi.fn().mockResolvedValueOnce(new Response(JSON.stringify({text: 'Cloud', duration: 2}))) + .mockRejectedValueOnce(new Error('secret')); + vi.stubGlobal('fetch', fetchMock); + const app = appWith(ai, usage); + const res = await post(app); + expect(res.status).toBe(200); + expect(await res.json()).toEqual({text: 'Cloud', durationMs: 2000}); + expect((await usage.report()).rows?.[0]).toMatchObject({provider: 'openai-compat', model: 'whisper-1', kind: 'transcribe', cost: null}); + const failed = await post(app); + expect(failed.status).toBe(502); + expect(await failed.text()).not.toContain('secret'); + expect((await usage.report()).rows).toHaveLength(1); + }); + + it('defaults to local, leaves it ungated, and gives explicit cloud configuration precedence', async () => { + const local = vi.fn(async () => new MockEngine()); + const service = new AiService(db, dir, local); + await store.updateInstanceConfig({ownerSubject: 'test#owner', guestAccess: 'write'}); + await store.setPageVisibility(pageId, 'public'); + expect((await post(appWith(service), undefined, true)).status).toBe(200); + expect(local).toHaveBeenCalledOnce(); + await service.setConfig({provider: 'off', transcription: {provider: 'openai-compat'}}); + expect((await service.transcriptionBackend()).provider).toBe('openai-compat'); + expect(local).toHaveBeenCalledOnce(); + }); + + it('returns identical 404s for missing, unreadable, unreferenced, and unrelated assets/pages', async () => { + await ai.setConfig({provider: 'mock'}); + await store.updateInstanceConfig({ownerSubject: 'test#owner', guestAccess: 'read'}); + await store.setPageVisibility(pageId, 'restricted'); + const app = appWith(); + const unreadable = await post(app, undefined, true); + expect(unreadable.status).toBe(404); + expect(await unreadable.json()).toEqual({error: 'asset not found'}); + const other = (await store.upsertPage({name: 'Other', data: snapshot()})).id; + for (const body of [{assetId: 'missing', pageId}, {assetId: '0'.repeat(64), pageId}, {assetId, pageId: 'missing'}, {assetId, pageId: '00000000-0000-0000-0000-000000000000'}, {assetId, pageId: other}]) { + expect((await post(app, body)).status).toBe(404); + } + await store.unrefAsset(assetId, pageId); + expect((await post(app)).status).toBe(404); + }); + + it('gives actionable 400s for off/unconfigured/local unavailable and rejects malformed bodies', async () => { + for (const provider of [undefined, 'off', 'local'] as const) { + await ai.setConfig({provider: 'off', ...(provider ? {transcription: {provider}} : {})}); + const res = await post(); + expect(res.status).toBe(400); + expect((await res.json()).error).toContain('Settings → AI'); + } + for (const body of [null, {}, {assetId: 42, pageId}, {assetId, pageId: ''}]) expect((await post(appWith(), body)).status).toBe(400); + }); + + it('uses multipart verbose JSON, auth, normalized URLs, exact bytes, and audio timing', async () => { + const fetchMock = vi.fn(async () => new Response(JSON.stringify({text: 'Hello', duration: 1.25, segments: [{start: 0, end: 1.25, text: 'Hello', id: 0}]}))); + vi.stubGlobal('fetch', fetchMock); + const engine = new OpenAiCompatEngine('https://audio.test/v1/', 'whisper', 'openai', 'private-key'); + const result = await engine.transcribe(new Uint8Array([0, 255, 3]), {mime: 'audio/webm'}); + expect(result).toEqual({text: 'Hello', durationMs: 1250, segments: [{start: 0, end: 1.25, text: 'Hello'}]}); + const [url, init] = fetchMock.mock.calls[0] as unknown as [string, RequestInit]; + expect(url).toBe('https://audio.test/v1/audio/transcriptions'); + expect(init.headers).toEqual({Authorization: 'Bearer private-key'}); + const form = init.body as FormData; + expect(form.get('model')).toBe('whisper'); + expect(form.get('response_format')).toBe('verbose_json'); + expect(form.get('timestamp_granularities[]')).toBe('segment'); + expect(new Uint8Array(await (form.get('file') as Blob).arrayBuffer())).toEqual(new Uint8Array([0, 255, 3])); + }); + + it('accepts text-only JSON and derives duration from valid segments; upstream failures are safe', async () => { + const engine = new OpenAiCompatEngine('https://audio.test', 'whisper'); + const fetchMock = vi.fn().mockResolvedValueOnce(new Response(JSON.stringify({text: 'Hi'}))) + .mockResolvedValueOnce(new Response(JSON.stringify({text: 'Hi', segments: [{start: 0, end: 2, text: 'Hi'}, {start: -1, end: 3, text: 'Bad'}]}))) + .mockResolvedValueOnce(new Response('secret', {status: 401})) + .mockResolvedValueOnce(new Response(JSON.stringify({text: 42}))); + vi.stubGlobal('fetch', fetchMock); + expect(await engine.transcribe(new Uint8Array())).toEqual({text: 'Hi', durationMs: 0}); + expect((await engine.transcribe(new Uint8Array())).durationMs).toBe(2000); + await expect(engine.transcribe(new Uint8Array())).rejects.toThrow('HTTP 401'); + await expect(engine.transcribe(new Uint8Array())).rejects.toThrow('Invalid transcription'); + }); + + it('HTTP client posts the contract and LocalDataClient rejects transcription', async () => { + const result = {text: 'Hello', durationMs: 1000}; + const fetchMock = vi.fn(async () => new Response(JSON.stringify(result), {headers: {'content-type': 'application/json'}})); + vi.stubGlobal('fetch', fetchMock); + expect(await new HttpDataClient('http://test').transcribeAsset(assetId, pageId)).toEqual(result); + const [url, init] = fetchMock.mock.calls[0] as unknown as [string, RequestInit]; + expect(url).toBe(`http://test${API.aiTranscribe}`); + expect(JSON.parse(init.body as string)).toEqual({assetId, pageId}); + await expect(LocalDataClient.prototype.transcribeAsset.call({aiUnavailable: () => new Error('AI unavailable')} as never)).rejects.toThrow('AI unavailable'); + }); +}); From 5c0497889afe14d6e713ad12e02d6a8c47448546 Mon Sep 17 00:00:00 2001 From: Eliot Lim Date: Sat, 3 Oct 2026 22:21:41 +0800 Subject: [PATCH 2/5] test(server): keep OIDC access-gate PAT fixture valid (MEET-2) --- packages/server/src/ableOidc.test.ts | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/packages/server/src/ableOidc.test.ts b/packages/server/src/ableOidc.test.ts index 13ee3512..a5328dc8 100644 --- a/packages/server/src/ableOidc.test.ts +++ b/packages/server/src/ableOidc.test.ts @@ -568,7 +568,8 @@ describe('able OIDC relying party', () => { issuer: 'local', scope: 'read', createdBy: 'test', - expiresAt: new Date(NOW + 60_000), + // PAT validation uses the database clock, not the injected OIDC clock. + expiresAt: new Date(Date.now() + 60_000), }); const app = createApp(store, undefined, new PageHub(), { ableOidc: {...idp.options, discoveryTtlMs: 0}, From fb162867fd1d4f9319da4de74984153531a523a8 Mon Sep 17 00:00:00 2001 From: Eliot Lim Date: Sat, 3 Oct 2026 23:16:10 +0800 Subject: [PATCH 3/5] docs: report MEET-2 implementation and verification blocker --- _report.md | 69 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 69 insertions(+) create mode 100644 _report.md diff --git a/_report.md b/_report.md new file mode 100644 index 00000000..8f38dca2 --- /dev/null +++ b/_report.md @@ -0,0 +1,69 @@ +MEET-2 implemented and committed. Full verification is BLOCKED by host disk exhaustion (`No space left on device`); it is NOT green. No push. The implementation provides request-scoped asset transcription, independent audio configuration, OpenAI-compatible multipart backend, SDK client, access/paid gates, write-only credentials, and usage attribution. + +Implementation commit: `69f01e0074451e11e58c370184649c24c1a6dbd2` (`feat(server,sdk): add asset transcription service and backend (MEET-2)`). Verification head: `5c0497889afe14d6e713ad12e02d6a8c47448546` (includes the OIDC fixture correction). The report is committed separately; final branch HEAD is reported in the handoff. + +Verification (all commands foreground): + +- MEET-2 focused suite: **10/10 passed**; server typecheck and changed-file lint passed. +- First `pnpm verify`: builds, generated-file check, all typechecks/lint, SDK **546**, UI **2319**, desktop **7**, and MCP contract tests passed. Server suite reported the expired OIDC fixture and mirror timeout; stopped that already-failed run after diagnosis. Server suite did not finish; end-to-end tests were not reached. +- Corrected OIDC fixture: **22/22 passed**. Mirror tests unchanged outside sandbox: **3/3 passed**. +- Second `pnpm verify` (outside sandbox): builds, generated-file check, all typechecks/lint passed; SDK money property test hit its five-second timeout under load (**545 passed, 1 timed out**). Unchanged SDK with `VITEST_MAX_WORKERS=1`: **546/546 passed**. +- Third `VITEST_MAX_WORKERS=1 pnpm verify` (outside sandbox): blocked rebuilding viewer by **ENOSPC**. `df -h .` reported **100% capacity, 133 MiB available** immediately afterward. No tests/assertions disabled, no timeouts increased. + +Relevant foreground output: + +```text +transcription.test.ts: Test Files 1 passed (1); Tests 10 passed (10) +ableOidc.test.ts: Test Files 1 passed (1); Tests 22 passed (22) +mirror.integration.test.ts (outside sandbox): Tests 3 passed (3) +SDK (VITEST_MAX_WORKERS=1): Test Files 32 passed (32); Tests 546 passed (546) + +Second full attempt: +FAIL src/money.test.ts > round-trips 1e6 seeded random amounts with zero drift +Error: Test timed out in 5000ms. + +Third full attempt: +Failed to write file in packages/ui/src/export/vendor/openbook-viewer.js +Caused by: No space left on device (os error 28) +ERR_PNPM_RECURSIVE_RUN_FIRST_FAIL @book.dev/ui build +Exit status 1 +``` + +Resume after freeing host disk capacity: rebuild viewer/server/MCP via the normal `build:libs` stage and rerun foreground `VITEST_MAX_WORKERS=1 pnpm verify` outside the sandbox. The last failed build may have left an incomplete ignored viewer artifact. Do not treat partial checks as full verification. Logs remain at `/tmp/meet2-verify.log`, `/tmp/meet2-verify-final.log`, and `/tmp/meet2-verify-serial.log`. + +Criterion → test map (`packages/server/src/transcription.test.ts`): + +| Criterion | Test | +| --- | --- | +| Config persistence, write-only keys, blank/omitted preserve, replacement, null clear | round-trips and redacts keys… | +| Deterministic mock; route success; usage row | returns the deterministic mock result… | +| Cloud paid gate independent of chat provider | denies cloud to claimed-instance guests… | +| Authenticated cloud, unknown cost, sanitized errors | allows authenticated cloud transcription… | +| Local default, ungated local, cloud precedence | defaults to local… | +| Missing/unreadable/unreferenced/unrelated assets, invalid/missing pages → 404 | returns identical 404s… | +| Off/unconfigured/local unavailable → actionable 400; malformed body | gives actionable 400s… | +| Multipart/auth/URL normalization/exact bytes/timing | uses multipart verbose JSON… | +| Text-only JSON, validated segments, duration fallback, provider failure | accepts text-only JSON… | +| HTTP SDK and LocalDataClient rejection | HTTP client posts the contract… | + +MEET-5 contract: + +- `DataClient.transcribeAsset(assetId: string, pageId: string): Promise`; HTTP `POST /api/ai/transcribe` with JSON `{assetId, pageId}`. Assets must already exist and reference that page; the caller must read both. Stored MIME is passed through, with no MIME allowlist dependency. +- Result `{text: string, segments?: Array<{start: number, end: number, text: string}>, durationMs: number}`. Segment offsets use **seconds**; duration uses **milliseconds**. Duration comes from the backend, otherwise maximum segment end, otherwise `0` (unknown). No queue, streaming, persistence, or page mutation. +- Errors: `400` missing arguments or disabled/unavailable local configuration (points to Settings → AI); `404` missing/unreadable/unrelated asset/page; `403` claimed-instance guest using cloud; `502` upstream failure with sanitized message. Existing application authentication/request gates still apply. +- `AiConfig.transcription?: {provider: 'off' | 'local' | 'openai-compat', baseUrl?: string, model?: string, apiKey?: string | null, apiKeySet?: boolean}`. Save through existing `aiSetConfig`, read through `aiStatus`. Missing section means local; `off` explicitly disables. Omitted section on save preserves prior audio settings. Present section replaces settings while preserving an omitted/blank key. Nonempty key replaces (trimmed); `null` clears. Responses remove keys and expose only `apiKeySet`; that flag is never persisted. +- Explicit `openai-compat` opts into the paid gate, even for a user-managed local URL. Defaults: `https://api.openai.com`, `whisper-1`. No chat key/config inheritance. Optional bearer key; base URL can include `/v1` and/or trailing slash. One multipart path requests `verbose_json` plus segment timestamps. JSON responses without segments are accepted. No automatic retry/downgrade on a provider rejecting verbose JSON. + +MEET-3 contract: + +- Implement `TranscriptionEngine` from `packages/server/src/ai/providers.ts`: `transcribe(bytes: Uint8Array, opts?: TranscribeOptions): Promise`. Options: `{filename?: string, mime?: string, signal?: AbortSignal}`. Respect cancellation; return the units above. +- Inject the optional third `AiService` constructor argument: `() => Promise`. Resolve/start the managed local backend lazily. Return `null` when unavailable. MEET-3 owns backend process/model lifecycle; MEET-2 does not dispose the injected engine per request. +- Reuse `new OpenAiCompatEngine(baseUrl, model)` for a local OpenAI-compatible server. The local resolver result is classified `local` and is ungated; omitted transcription config selects this resolver. `transcriptionBackend()` captures one backend/provider/model per request before the paid gate. Explicit `off` disables; explicit cloud wins; local resolver comes next; chat `mock` supplies deterministic fallback for tests/demos; otherwise actionable `400`. +- `AiEngine.transcribe` is optional, so chat-only engines need no changes. The endpoint never calls chat readiness/model probes. +- Successful requests log one `kind: 'transcribe'` row under `openai-compat`, `local`, or `mock` with server-resolved principal/model. Audio token counts are `0` (unreported); cloud cost is `null`, local/mock cost `0`. No audio-minute pricing/schema migration. The local model label is configured `transcription.model` or `local`. + +Verification follow-up: the first sandboxed run found an existing OIDC PAT fixture expired on September 8, 2026. Commit `5c049788` makes its expiry relative to the real database clock, preserving every assertion; all 22 OIDC tests passed afterward. Existing mirror integration tests also hit filesystem-watcher `EMFILE` errors and a timeout in the sandbox. All 3 passed unchanged outside the sandbox. Stopped the already-failed sandboxed run and restarted full foreground verification outside the sandbox. No tests or assertions disabled, no timeout increased. + +Deviations/limitations: Local implementation remains the planned MEET-3 hook; no MEET-1 MIME changes. Cloud errors do not trigger a silent local fallback. Tests use mocked upstream fetch, not a live paid service. Backend timeout is five minutes. No existing tests deleted or weakened. + +Open blocker: host disk capacity; full verification and end-to-end checks remain outstanding. No interface questions. MEET-3 should wire its resolver at AiService construction; MEET-5 should render `apiKeySet` and surface the actionable errors. From 79f576aa721778b5b6d5e9c7c4b1fe647fc33cea Mon Sep 17 00:00:00 2001 From: Eliot Lim Date: Sun, 4 Oct 2026 00:45:38 +0800 Subject: [PATCH 4/5] docs: record green serialized verification for MEET-2 --- _report.md | 56 +++++++++++++++++++++++++++++------------------------- 1 file changed, 30 insertions(+), 26 deletions(-) diff --git a/_report.md b/_report.md index 8f38dca2..b58b08c7 100644 --- a/_report.md +++ b/_report.md @@ -1,35 +1,39 @@ -MEET-2 implemented and committed. Full verification is BLOCKED by host disk exhaustion (`No space left on device`); it is NOT green. No push. The implementation provides request-scoped asset transcription, independent audio configuration, OpenAI-compatible multipart backend, SDK client, access/paid gates, write-only credentials, and usage attribution. +MEET-2 PASS. Implemented and committed; full foreground verification is GREEN. No push. Request-scoped asset transcription, independent audio configuration, OpenAI-compatible multipart backend, SDK client, access/paid gates, write-only credentials, and usage attribution are complete. -Implementation commit: `69f01e0074451e11e58c370184649c24c1a6dbd2` (`feat(server,sdk): add asset transcription service and backend (MEET-2)`). Verification head: `5c0497889afe14d6e713ad12e02d6a8c47448546` (includes the OIDC fixture correction). The report is committed separately; final branch HEAD is reported in the handoff. +Implementation commit: `69f01e0074451e11e58c370184649c24c1a6dbd2`. OIDC fixture correction: `5c0497889afe14d6e713ad12e02d6a8c47448546`. Verified head: `fb162867fd1d4f9319da4de74984153531a523a8`; this report-only update is committed afterward, with final branch HEAD supplied in the handoff. -Verification (all commands foreground): +Final verification: `VITEST_MAX_WORKERS=1 pnpm verify`, foreground, outside the sandbox, completed **2026-10-04 (Asia/Singapore), exit 0**. Single-worker execution was requested by the manager after confirming the unchanged SDK money test's host-load timeout. No test assertions, counts, or timeouts changed. -- MEET-2 focused suite: **10/10 passed**; server typecheck and changed-file lint passed. -- First `pnpm verify`: builds, generated-file check, all typechecks/lint, SDK **546**, UI **2319**, desktop **7**, and MCP contract tests passed. Server suite reported the expired OIDC fixture and mirror timeout; stopped that already-failed run after diagnosis. Server suite did not finish; end-to-end tests were not reached. -- Corrected OIDC fixture: **22/22 passed**. Mirror tests unchanged outside sandbox: **3/3 passed**. -- Second `pnpm verify` (outside sandbox): builds, generated-file check, all typechecks/lint passed; SDK money property test hit its five-second timeout under load (**545 passed, 1 timed out**). Unchanged SDK with `VITEST_MAX_WORKERS=1`: **546/546 passed**. -- Third `VITEST_MAX_WORKERS=1 pnpm verify` (outside sandbox): blocked rebuilding viewer by **ENOSPC**. `df -h .` reported **100% capacity, 133 MiB available** immediately afterward. No tests/assertions disabled, no timeouts increased. - -Relevant foreground output: +| Final stage | Result | +| --- | --- | +| ESLint rule tests | 6 passed | +| SDK/UI/MCP/server builds; generated-file check | Passed | +| All workspace typechecks and lint | Passed | +| SDK | 32 files; 546 tests passed | +| UI | 244 files; 2,319 tests passed | +| Desktop | 2 files; 7 tests passed | +| Server (includes MEET-2 tests) | 98 files; 1,336 tests passed, 6 skipped | +| Vitest total | 376 files; 4,208 passed, 6 skipped | +| MCP contract scripts | All passed: listed 3, pages 17, suggestions 44, databases 17, assets 13, blocks 80, tables 54, forms 40, block types 60, endpoint 10; README covers 50 tools; coverage includes 44 catalogue types + 9 plugin blocks | +| Server end-to-end | 256/256 checks passed | +| MCP end-to-end | 70/70 checks passed | + +Final foreground output excerpt: ```text -transcription.test.ts: Test Files 1 passed (1); Tests 10 passed (10) -ableOidc.test.ts: Test Files 1 passed (1); Tests 22 passed (22) -mirror.integration.test.ts (outside sandbox): Tests 3 passed (3) -SDK (VITEST_MAX_WORKERS=1): Test Files 32 passed (32); Tests 546 passed (546) - -Second full attempt: -FAIL src/money.test.ts > round-trips 1e6 seeded random amounts with zero drift -Error: Test timed out in 5000ms. - -Third full attempt: -Failed to write file in packages/ui/src/export/vendor/openbook-viewer.js -Caused by: No space left on device (os error 28) -ERR_PNPM_RECURSIVE_RUN_FIRST_FAIL @book.dev/ui build -Exit status 1 +packages/sdk test: Test Files 32 passed (32) +packages/sdk test: Tests 546 passed (546) +packages/ui test: Test Files 244 passed (244) +packages/ui test: Tests 2319 passed (2319) +packages/app test: Test Files 2 passed (2) +packages/app test: Tests 7 passed (7) +packages/server test: Test Files 98 passed (98) +packages/server test: Tests 1336 passed | 6 skipped (1342) +✅ ALL 256 CHECKS PASSED — embedded, persistence, headless, trash-cleanup, access-token, and ledger flows verified. +✅ ALL 70 CHECKS PASSED — MCP handshake, catalogue, and every tool verified. ``` -Resume after freeing host disk capacity: rebuild viewer/server/MCP via the normal `build:libs` stage and rerun foreground `VITEST_MAX_WORKERS=1 pnpm verify` outside the sandbox. The last failed build may have left an incomplete ignored viewer artifact. Do not treat partial checks as full verification. Logs remain at `/tmp/meet2-verify.log`, `/tmp/meet2-verify-final.log`, and `/tmp/meet2-verify-serial.log`. +Full log: `/tmp/meet2-manager-serial-verify.log`. Focused MEET-2 suite previously passed 10/10. Earlier disk exhaustion and the unchanged money-test timeout are resolved for this final run; the viewer/server/MCP bundles were rebuilt successfully. No tests were disabled or weakened. Criterion → test map (`packages/server/src/transcription.test.ts`): @@ -66,4 +70,4 @@ Verification follow-up: the first sandboxed run found an existing OIDC PAT fixtu Deviations/limitations: Local implementation remains the planned MEET-3 hook; no MEET-1 MIME changes. Cloud errors do not trigger a silent local fallback. Tests use mocked upstream fetch, not a live paid service. Backend timeout is five minutes. No existing tests deleted or weakened. -Open blocker: host disk capacity; full verification and end-to-end checks remain outstanding. No interface questions. MEET-3 should wire its resolver at AiService construction; MEET-5 should render `apiKeySet` and surface the actionable errors. +Open blockers: none. Full verification and end-to-end checks passed. No interface questions. MEET-3 should wire its resolver at AiService construction; MEET-5 should render `apiKeySet` and surface the actionable errors. From bbccdbc1d976fb722a8372e82604a3dc31a31b2c Mon Sep 17 00:00:00 2001 From: Eliot Lim Date: Sun, 4 Oct 2026 00:51:23 +0800 Subject: [PATCH 5/5] chore: drop worker artifacts (MEET-2) --- _brief.md | 29 ---------------------- _report.md | 73 ------------------------------------------------------ 2 files changed, 102 deletions(-) delete mode 100644 _brief.md delete mode 100644 _report.md diff --git a/_brief.md b/_brief.md deleted file mode 100644 index eac54b32..00000000 --- a/_brief.md +++ /dev/null @@ -1,29 +0,0 @@ -# MEET-2 — Transcription service + OpenAI-compatible backend - -You are a Worker agent on the OpenBook team. Work ONLY in this worktree (`/Users/eliot/Workspaces/OpenBook-wt-meet-2`, branch `feat/meet-2-transcribe`). Do NOT push. Do NOT touch main. Conventional commits (`feat(server,sdk): … (MEET-2)`), committed incrementally. - -## Task -Add a transcription capability to the AI layer. No engine has audio today. Key anchors (recon-time line hints — trust the code): -- Engines: `packages/server/src/ai/providers.ts` — `MockEngine` :108, `OpenAiCompatEngine` :196, `AnthropicEngine` :620, factory `createEngine` :821. -- Config: `AiConfig` `packages/sdk/src/ai.ts:66`, persisted in DB `settings` key `'ai'` (`ai/service.ts:115-123`); API keys are write-only over the wire (`resolveKey` service.ts:31-37 — blank keeps, null clears). Preserve those semantics for any new key field. -- Routes: `packages/server/src/ai/routes.ts` — `aiComplete` :137 is the request-scoped shape to mirror; paid-provider gate `requirePaidInferenceAccess` :412; usage logging `ai/usage.ts`. -- Access: default-deny via `access.ts` (`requireAccess`); unreadable → 404. - -## Design (contract for MEET-3 local whisper and MEET-5 UI — keep the interface clean) -1. Extend `AiConfig` with a `transcription` section: `{ provider: 'off' | 'local' | 'openai-compat', baseUrl?, model?, apiKey? }`. **Resolution order: explicit cloud config > local (arrives in MEET-3; stub the enum/dispatch now) > clear actionable error pointing at Settings → AI.** Local is the product default — cloud only when explicitly configured. -2. New `transcribe` capability speaking the OpenAI-compatible `POST /v1/audio/transcriptions` multipart shape (works for OpenAI cloud and local servers like faster-whisper/speaches — ONE code path). Request `verbose_json`/segments where available. -3. Server route (under ai/routes.ts): POST taking `{ assetId, pageId }` (bytes already in the asset store — fetch via the store with the caller's read access enforced), returning `{ text, segments?: [{start, end, text}], durationMs }`. Request-scoped like aiComplete. Do NOT build a job queue. -4. Cloud transcription goes through `requirePaidInferenceAccess`; `local` will be ungated. Log usage rows. -5. sdk: `transcribeAsset(assetId, pageId)` on the HTTP client (pattern near `client.ts:1571`). `LocalDataClient` (`server/localClient.ts:888-932`) keeps rejecting AI calls — extend its rejection list to the new method. - -## Acceptance (each maps to a test) -- MockEngine grows a deterministic `transcribe` for tests. -- Unit tests: config round-trip incl. write-only key semantics; route happy path via mock; paid-gate enforced for cloud provider; 404 on missing/unreadable asset; `off`/unconfigured → actionable 4xx error. -- NOTE: audio MIME allowlist work is MEET-1 (parallel branch) — do not depend on it; transcribe takes the asset bytes regardless of stored MIME. - -## Definition of done -- `pnpm verify` green FOREGROUND in this worktree, output in report. (Provisioned; artifact-failure symptom → rebuild viewer/server/mcp bundles.) -- All committed, not pushed. Write `_report.md` (outcome first, head sha, criterion→test map, interface summary for MEET-3/MEET-5 consumers, deviations, open questions) and reply with it. Terse. - -## Rules -Never poll external state. Stuck after a real attempt → commit + report where wedged. Never delete/weaken existing tests (reviewers check the commit RANGE). diff --git a/_report.md b/_report.md deleted file mode 100644 index b58b08c7..00000000 --- a/_report.md +++ /dev/null @@ -1,73 +0,0 @@ -MEET-2 PASS. Implemented and committed; full foreground verification is GREEN. No push. Request-scoped asset transcription, independent audio configuration, OpenAI-compatible multipart backend, SDK client, access/paid gates, write-only credentials, and usage attribution are complete. - -Implementation commit: `69f01e0074451e11e58c370184649c24c1a6dbd2`. OIDC fixture correction: `5c0497889afe14d6e713ad12e02d6a8c47448546`. Verified head: `fb162867fd1d4f9319da4de74984153531a523a8`; this report-only update is committed afterward, with final branch HEAD supplied in the handoff. - -Final verification: `VITEST_MAX_WORKERS=1 pnpm verify`, foreground, outside the sandbox, completed **2026-10-04 (Asia/Singapore), exit 0**. Single-worker execution was requested by the manager after confirming the unchanged SDK money test's host-load timeout. No test assertions, counts, or timeouts changed. - -| Final stage | Result | -| --- | --- | -| ESLint rule tests | 6 passed | -| SDK/UI/MCP/server builds; generated-file check | Passed | -| All workspace typechecks and lint | Passed | -| SDK | 32 files; 546 tests passed | -| UI | 244 files; 2,319 tests passed | -| Desktop | 2 files; 7 tests passed | -| Server (includes MEET-2 tests) | 98 files; 1,336 tests passed, 6 skipped | -| Vitest total | 376 files; 4,208 passed, 6 skipped | -| MCP contract scripts | All passed: listed 3, pages 17, suggestions 44, databases 17, assets 13, blocks 80, tables 54, forms 40, block types 60, endpoint 10; README covers 50 tools; coverage includes 44 catalogue types + 9 plugin blocks | -| Server end-to-end | 256/256 checks passed | -| MCP end-to-end | 70/70 checks passed | - -Final foreground output excerpt: - -```text -packages/sdk test: Test Files 32 passed (32) -packages/sdk test: Tests 546 passed (546) -packages/ui test: Test Files 244 passed (244) -packages/ui test: Tests 2319 passed (2319) -packages/app test: Test Files 2 passed (2) -packages/app test: Tests 7 passed (7) -packages/server test: Test Files 98 passed (98) -packages/server test: Tests 1336 passed | 6 skipped (1342) -✅ ALL 256 CHECKS PASSED — embedded, persistence, headless, trash-cleanup, access-token, and ledger flows verified. -✅ ALL 70 CHECKS PASSED — MCP handshake, catalogue, and every tool verified. -``` - -Full log: `/tmp/meet2-manager-serial-verify.log`. Focused MEET-2 suite previously passed 10/10. Earlier disk exhaustion and the unchanged money-test timeout are resolved for this final run; the viewer/server/MCP bundles were rebuilt successfully. No tests were disabled or weakened. - -Criterion → test map (`packages/server/src/transcription.test.ts`): - -| Criterion | Test | -| --- | --- | -| Config persistence, write-only keys, blank/omitted preserve, replacement, null clear | round-trips and redacts keys… | -| Deterministic mock; route success; usage row | returns the deterministic mock result… | -| Cloud paid gate independent of chat provider | denies cloud to claimed-instance guests… | -| Authenticated cloud, unknown cost, sanitized errors | allows authenticated cloud transcription… | -| Local default, ungated local, cloud precedence | defaults to local… | -| Missing/unreadable/unreferenced/unrelated assets, invalid/missing pages → 404 | returns identical 404s… | -| Off/unconfigured/local unavailable → actionable 400; malformed body | gives actionable 400s… | -| Multipart/auth/URL normalization/exact bytes/timing | uses multipart verbose JSON… | -| Text-only JSON, validated segments, duration fallback, provider failure | accepts text-only JSON… | -| HTTP SDK and LocalDataClient rejection | HTTP client posts the contract… | - -MEET-5 contract: - -- `DataClient.transcribeAsset(assetId: string, pageId: string): Promise`; HTTP `POST /api/ai/transcribe` with JSON `{assetId, pageId}`. Assets must already exist and reference that page; the caller must read both. Stored MIME is passed through, with no MIME allowlist dependency. -- Result `{text: string, segments?: Array<{start: number, end: number, text: string}>, durationMs: number}`. Segment offsets use **seconds**; duration uses **milliseconds**. Duration comes from the backend, otherwise maximum segment end, otherwise `0` (unknown). No queue, streaming, persistence, or page mutation. -- Errors: `400` missing arguments or disabled/unavailable local configuration (points to Settings → AI); `404` missing/unreadable/unrelated asset/page; `403` claimed-instance guest using cloud; `502` upstream failure with sanitized message. Existing application authentication/request gates still apply. -- `AiConfig.transcription?: {provider: 'off' | 'local' | 'openai-compat', baseUrl?: string, model?: string, apiKey?: string | null, apiKeySet?: boolean}`. Save through existing `aiSetConfig`, read through `aiStatus`. Missing section means local; `off` explicitly disables. Omitted section on save preserves prior audio settings. Present section replaces settings while preserving an omitted/blank key. Nonempty key replaces (trimmed); `null` clears. Responses remove keys and expose only `apiKeySet`; that flag is never persisted. -- Explicit `openai-compat` opts into the paid gate, even for a user-managed local URL. Defaults: `https://api.openai.com`, `whisper-1`. No chat key/config inheritance. Optional bearer key; base URL can include `/v1` and/or trailing slash. One multipart path requests `verbose_json` plus segment timestamps. JSON responses without segments are accepted. No automatic retry/downgrade on a provider rejecting verbose JSON. - -MEET-3 contract: - -- Implement `TranscriptionEngine` from `packages/server/src/ai/providers.ts`: `transcribe(bytes: Uint8Array, opts?: TranscribeOptions): Promise`. Options: `{filename?: string, mime?: string, signal?: AbortSignal}`. Respect cancellation; return the units above. -- Inject the optional third `AiService` constructor argument: `() => Promise`. Resolve/start the managed local backend lazily. Return `null` when unavailable. MEET-3 owns backend process/model lifecycle; MEET-2 does not dispose the injected engine per request. -- Reuse `new OpenAiCompatEngine(baseUrl, model)` for a local OpenAI-compatible server. The local resolver result is classified `local` and is ungated; omitted transcription config selects this resolver. `transcriptionBackend()` captures one backend/provider/model per request before the paid gate. Explicit `off` disables; explicit cloud wins; local resolver comes next; chat `mock` supplies deterministic fallback for tests/demos; otherwise actionable `400`. -- `AiEngine.transcribe` is optional, so chat-only engines need no changes. The endpoint never calls chat readiness/model probes. -- Successful requests log one `kind: 'transcribe'` row under `openai-compat`, `local`, or `mock` with server-resolved principal/model. Audio token counts are `0` (unreported); cloud cost is `null`, local/mock cost `0`. No audio-minute pricing/schema migration. The local model label is configured `transcription.model` or `local`. - -Verification follow-up: the first sandboxed run found an existing OIDC PAT fixture expired on September 8, 2026. Commit `5c049788` makes its expiry relative to the real database clock, preserving every assertion; all 22 OIDC tests passed afterward. Existing mirror integration tests also hit filesystem-watcher `EMFILE` errors and a timeout in the sandbox. All 3 passed unchanged outside the sandbox. Stopped the already-failed sandboxed run and restarted full foreground verification outside the sandbox. No tests or assertions disabled, no timeout increased. - -Deviations/limitations: Local implementation remains the planned MEET-3 hook; no MEET-1 MIME changes. Cloud errors do not trigger a silent local fallback. Tests use mocked upstream fetch, not a live paid service. Backend timeout is five minutes. No existing tests deleted or weakened. - -Open blockers: none. Full verification and end-to-end checks passed. No interface questions. MEET-3 should wire its resolver at AiService construction; MEET-5 should render `apiKeySet` and surface the actionable errors.