diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index db88b07b..453fd80a 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -309,6 +309,38 @@ The FORM-7 MCP surface in `packages/mcp/src/server.ts` provides `list_forms`, through the resolved per-page agent-edits policy (suggest by default), and key regeneration is intentionally author-UI-only. +### Meeting blocks + +Meetings are native `type:'meeting'` container blocks registered by +`packages/ui/src/blockeditor/MeetingBlockView.tsx`. The +[representation contract](docs/meeting-block.md) defines audio asset references, +timestamped transcript segments, summary, title, and status props. Manual notes +remain ordinary CRDT child blocks; generated transcript and summary are prop +snapshots. The user workflow is in [meeting notes](docs/meeting-notes.md). + +`MeetingRecorder` restarts MediaRecorder every 45 seconds and on pause/resume. +Each uploaded chunk has its own container header, unlike recorder timeslices, +so playback, retries, transcription, and export work on standalone files. +Uploads use page-associated assets; `POST /api/ai/transcribe` receives the asset +and page IDs, enforces access, and records usage. Completed chunks append +transcript segments progressively. Summary generation is an explicit streamed +AI request; cancellation preserves the previous durable summary. + +Transcription resolves separately from chat: explicit off rejects; an explicit +OpenAI-compatible transcription provider opts into that endpoint; otherwise the +local resolver runs, followed by the deterministic mock fallback only when chat +provider is mock. An unavailable local engine returns a configuration error, +never implicit cloud fallback. Local Whisper transcription ships by default; **Settings → AI** provides the +model download. See [local transcription setup](docs/local-transcription.md) for +Whisper and FFmpeg installation and runtime requirements. + +Audio export downloads a single original file or an ordered timestamped ZIP of +chunks, without remuxing or deleting library assets. Markdown and HTML exports +include transcript, summary, and child notes with audio references; audio bytes +are exported separately. The browser epic proof is +`packages/web/e2e/meeting-epic.spec.ts`, using a WebAudio microphone substitute +with real recording, asset, transcription, generation, and download paths. + ### Optional local AI (`packages/server/src/ai/`) An opt-in, local-only model subsystem (Settings → AI). Pluggable engines diff --git a/docs/audits/block-api-coverage.md b/docs/audits/block-api-coverage.md index 615fa00b..7a19438c 100644 --- a/docs/audits/block-api-coverage.md +++ b/docs/audits/block-api-coverage.md @@ -46,6 +46,7 @@ | tooltipcard | ✅ | ✅ | ➖ | ✅ | ➖ | ➖ | | dbview | ✅ | ✅ | ➖ | ✅ | ➖ | ➖ | | dbform | ✅ | ✅ | ➖ | ✅ | ➖ | ➖ | +| meeting | ✅ | ✅ | ➖ | ✅ | ➖ | ➖ | | form | ✅ | ✅ | ➖ | ✅ | ➖ | ➖ | | openbook.ledger/journal-entry | ✅ | ✅ | ➖ | ✅ | ➖ | ✅ | | openbook.ledger/trial-balance | ✅ | ✅ | ➖ | ✅ | ➖ | ✅ | diff --git a/docs/local-transcription.md b/docs/local-transcription.md new file mode 100644 index 00000000..6accd3a2 --- /dev/null +++ b/docs/local-transcription.md @@ -0,0 +1,44 @@ +# Local transcription + +OpenBook transcribes recordings locally by default, without a cloud API key. The server uses the optional whisper.cpp `whisper-cli` executable and FFmpeg. Neither executable nor model weights are bundled with OpenBook; normal CI skips native inference explicitly. + +## Setup + +1. Install whisper.cpp and FFmpeg on the server host. Follow the [whisper.cpp build instructions](https://github.com/ggml-org/whisper.cpp#quick-start) or use your host package manager. +2. Put `whisper-cli` and `ffmpeg` on the server process's PATH. Alternatively, set executable paths before starting OpenBook: + + ```sh + export OPENBOOK_WHISPER_BIN=/absolute/path/to/whisper-cli + export OPENBOOK_FFMPEG_BIN=/absolute/path/to/ffmpeg + ``` + +3. In **Settings → AI**, select **Download Whisper base**. This uses the existing authenticated model download and progress flow. Downloads go to `OPENBOOK_MODELS_DIR` when set, otherwise the server data directory's `models` folder (or `~/.openbook/models` without a data directory). The completed model is discovered without restarting. +4. Check that Settings reports the runtime and model ready, then transcribe a recording. + +The default model is multilingual Whisper base (`ggml-base.bin`, approximately 142 MiB), with automatic language detection. It trades some accuracy on noisy speech, accents, and difficult multilingual recordings for a smaller download and lower memory use than larger models. Whisper downloads do not change the selected chat model. + +Explicit cloud transcription configuration takes precedence over local inference. Otherwise the server tries local transcription, then the existing mock fallback. Local transcription works even with chat disabled; explicitly disabling transcription still disables it. Missing executables or model weights produce an actionable error pointing to Settings → AI. + +## Processing and limits + +FFmpeg converts recordings to mono 16 kHz PCM WAV. Whisper loads the model for each job and returns text plus segments in seconds; `durationMs` is the rounded maximum segment endpoint in milliseconds. Each job uses a private temporary directory. Cancellation and server shutdown kill active child processes, wait for them to close, and remove scratch files. + +Each server permits at most **two local transcription jobs at once**, shared across all clients. There is no queue. Busy requests return HTTP **429** with `Retry-After: 5`. Permits remain held through temporary-file cleanup and are released on success, failure, or cancellation. + +The transcription route also allows **six local requests per socket IP per 60-second fixed window**. Excess requests return HTTP **429** with `Retry-After: 60`. Client-supplied forwarding headers do not change this key; clients behind a reverse proxy may share its socket IP budget. The limit applies only when the resolved backend is local; cloud and mock backends are unaffected. + +## Native smoke test + +Install the optional executables and download `ggml-base.bin` into your model directory, then run from the repository root: + +```sh +OPENBOOK_TEST_WHISPER=1 \ +OPENBOOK_MODELS_DIR=/absolute/path/to/models \ +OPENBOOK_WHISPER_BIN=/absolute/path/to/whisper-cli \ +OPENBOOK_FFMPEG_BIN=/absolute/path/to/ffmpeg \ +VITEST_MAX_WORKERS=1 \ +pnpm --filter @book.dev/server exec vitest run src/transcription.test.ts \ + -t 'native whisper' +``` + +This opt-in test synthesizes a small WAV and exercises the HTTP transcription route without cloud keys, asserting the result shape. It is a runtime integration check, not a speech-accuracy benchmark. Without `OPENBOOK_TEST_WHISPER=1`, the native test is explicitly skipped. The regular subprocess-fixture tests cover concurrency, cancellation, failures, timestamp conversion, and cleanup without installing native dependencies. diff --git a/docs/meeting-block.md b/docs/meeting-block.md new file mode 100644 index 00000000..4e52a824 --- /dev/null +++ b/docs/meeting-block.md @@ -0,0 +1,44 @@ +# Meeting block — THE REPRESENTATION CONTRACT (MEET-4) + +`meeting` is a `kit` block with `nature: 'container'`, no required parent, and +`kitValue: false`. It publishes no reactive value. Its `children` are ordinary +manual-note blocks (paragraphs, todos, headings, groups, etc.), not transcript +segments. No dedicated notes slot or child-only type is required. + +All top-level props are optional. Missing status means `idle`; missing arrays +mean empty lists; missing summary/title mean empty text; missing startedAt means +not started. These are consumer defaults, not schema-inserted persisted values. +As with other block props, patches shallow-merge; `null` removes a top-level key. +Unknown top-level props remain allowed for forward compatibility. + +| Prop | Stored shape and meaning | +| --- | --- | +| `status` | `'idle' \| 'recording' \| 'processing' \| 'done'`. Validation checks the enum, not state transitions. | +| `audioChunks` | Ordered array of `{assetId: string, durationMs: number, startedAtMs?: number}`. `assetId` is a nonempty asset reference (max 512 characters), never inline audio. `durationMs` is the chunk duration; optional `startedAtMs` is an offset from meeting start, **not** Unix time. Array order is capture/playback order. | +| `transcript` | Ordered array of `{startMs: number, endMs: number, text: string}`. Offsets are relative to meeting start; `endMs >= startMs`. Text is plain text. Array order is display order; overlapping segments are allowed. | +| `summary` | Plain string; no rich-text runs or Markdown interpretation is required. | +| `startedAt` | Unix epoch milliseconds, a finite nonnegative number. | +| `title` | Optional plain string. | + +All durations and offsets are finite nonnegative numbers in milliseconds; +fractional milliseconds are accepted. Every listed nested field is required +except `startedAtMs`; nested objects reject extra keys and null fields. Empty +arrays and empty transcript text are valid. Cross-field `endMs >= startMs` is +checked at runtime and described in the JSON schema (standard JSON Schema cannot +express a comparison to a sibling field). Asset existence, timeline sorting, +status transitions, and transcription size limits are producer responsibilities. + +Audio chunks and transcript segments use structured props, following existing +kit option/rich-run arrays and the form's structured schema prop. They are +machine-produced snapshots, while manual notes need independently editable CRDT +children. Each array is one prop value: updates replace the whole array, with +no per-segment merge guarantee. Recording/transcription producers should batch +updates and serialize writes to avoid concurrent array replacement losing data. +This keeps the first representation small and consistent; large-transcript +storage migration, if needed later, must explicitly version this contract. + +MEET-4 registers a minimal meeting shell through `registerArtifactKit` in both +editor and viewer hosts. It displays title, status, and saved note-block count; +notes remain stored but are not yet rendered/editable inside the shell. It has +no slash-menu entry or recording controls. MEET-5 replaces this renderer and +must render the existing child blocks rather than migrate them into props. diff --git a/docs/meeting-notes.md b/docs/meeting-notes.md new file mode 100644 index 00000000..ed33a2ec --- /dev/null +++ b/docs/meeting-notes.md @@ -0,0 +1,63 @@ +# Meeting notes + +On a saved page, type `/meeting` and choose **Meeting**. Keep the page connected +to its OpenBook server so recordings can be saved to the library. + +## Record and transcribe + +Click **Record** and allow microphone access. OpenBook saves roughly 45-second +chunks and transcribes each completed upload; transcript lines appear as chunks +finish, without waiting for the entire meeting. Each chunk is a standalone audio +file, so it can be played or exported independently. + +**Pause** closes the current chunk and silences capture. **Resume** starts a new +chunk; paused time is excluded from the recording timeline. **Stop** finishes +capture and lets pending uploads and transcription settle. Keep the page open +until processing completes. Failed uploads retain an in-session audio copy with +save/retry controls; transcription failures preserve uploaded audio and offer +**Retry transcription**. Leaving the page can lose audio that has not uploaded. + +Local Whisper transcription ships by default, independently of the chat model. +In **Settings → AI**, select **Download Whisper base** to download the model. +See [local transcription](local-transcription.md) for the required Whisper and +FFmpeg installation, model setup, and runtime checks. If local transcription +is unavailable, recording and manual notes still work; OpenBook does not silently +fall back to a cloud service. Cloud transcription requires explicit opt-in in +**Settings → AI**. Selecting a cloud chat model alone does not opt audio into +cloud transcription. + +## Summaries and notes + +Configure a generation provider in **Settings → AI**, then click **Generate +summary** once transcript text exists. Summaries never start automatically. +Text appears while generation streams. **Cancel** discards the unfinished +replacement and retains any previous summary. **Regenerate summary** requests a +new version. Long transcripts are clipped to the first 4,000 characters for the +summary prompt, with an instruction to disclose that it covers an excerpt. + +Type directly under **Notes**, including while recording. Notes are ordinary +editable blocks inside the meeting, independent of generated transcript and +summary text. + +## Export and privacy + +Click **Export audio** to download one audio file for a single saved chunk, or a +ZIP containing ordered, timestamped files for multiple chunks. Files retain their +original formats and bytes; the ZIP does not join or re-encode the recording. +Export leaves the original audio in your library. Audio is stored with the page's +assets and follows library/page access rules; it is not uploaded to a third-party +transcription service unless you explicitly configure one. With a remote library, +capture uploads to that library's server, so “local transcription” refers to +processing on the server rather than necessarily on the microphone's device. + +Use **Page actions → Export → Markdown (.md)** for transcript timestamps, +summary, manual notes, and audio references. Markdown is not an audio backup; +use **Export audio** for the actual recordings. + +## Desktop + +Allow microphone access for OpenBook in your OS privacy settings and reopen the +app if permission changes require it. Browser permission instructions may still +appear in the shared UI. Follow the [desktop microphone manual checks](../packages/app/README.md#meeting-microphone-manual-checks) +to verify the actual webview and OS permission behavior; Chromium e2e coverage +uses synthetic audio and cannot establish desktop microphone support. diff --git a/packages/app/README.md b/packages/app/README.md index f96da3dd..8d10c9ef 100644 --- a/packages/app/README.md +++ b/packages/app/README.md @@ -4,6 +4,49 @@ From the workspace root, run it with `pnpm tauri dev`. Create a desktop build with `pnpm build:desktop`. +## Meeting microphone (macOS) + +`src-tauri/Info.plist` is merged into the bundle by Tauri and supplies +`NSMicrophoneUsageDescription`. The existing hardened-runtime entitlements file +now includes `com.apple.security.device.audio-input`. Recording is requested only +from the meeting block's Record action, via `getUserMedia({audio: true})`. + +The locked Tauri 2.11.2 / Wry 0.55.1 already installs the Rust +[`WKUIDelegate` media-capture callback](https://github.com/tauri-apps/wry/blob/wry-v0.55.1/src/wkwebview/class/wry_web_view_ui_delegate.rs#L126-L137). +It grants the webview permission layer; macOS TCC still controls device access, +prompts for initial consent, and persists Allow/Deny. Do not replace Wry's UI +delegate or add a second permission store. This version does not expose the newer +`with_permission_handler` API. An OS denial reaches the recorder's existing +graceful microphone error and returns it to idle. + +The recorder restarts every 45 seconds for independently playable files, targeting +64 kbit/s: approximately 360,000 bytes, or 480,000 base64 characters per chunk +(container overhead and actual encoder bitrate vary). Safari can fall back to +`audio/mp4`; the server accepts both MP4 and WebM. SDK uploads and playback use +base64 JSON across `tauriFetch`, below the default 10 MiB decoded asset cap and +the server's `ceil(cap * 4 / 3) + 64 KiB` request cap at this target size. The +desktop transport test checks two such chunks byte-for-byte with mocked native +IPC; it does not prove native capture or audible playback. + +Manual release-app check (requires an interactive macOS microphone): + +1. Run `pnpm build:desktop`, quit any other OpenBook instance, then launch + `packages/app/src-tauri/target/release/bundle/macos/OpenBook.app/Contents/MacOS/OpenBook`. + Use a release bundle: `tauri dev` does not launch the managed server sidecar. +2. Insert a meeting block on a page, press Record, and allow the macOS prompt. + Record for more than 45 seconds, then Stop. Confirm two uploaded audio chunks + and play each; reload the page and play again to verify sidecar persistence. +3. Restart the same app and record again; the OS should retain consent. +4. Disable OpenBook under System Settings → Privacy & Security → Microphone, + relaunch, and press Record. Confirm the microphone error appears, the block + returns to idle, and no capture/upload starts. Re-enable permission to recover. + +Distribution must rebuild, sign, and notarize the app with the new entitlement +and purpose string through the existing release process. No signing identity, +hardened-runtime setting, or notarization configuration is changed here. Permission +continuity across locally signed and distributed builds needs a check using the +owner's normal signing identity; signing changes remain owner-gated. + ## Sidecar supervision verification The bundled sidecar is supervised only in a release build (`tauri dev` uses the @@ -58,3 +101,25 @@ a fresh bounded run. Crash-loop exhaustion, the 1/2/4/8/16-second bound, healthy reset, deliberate stop suppression, and repair reset are deterministic unit tests in `src-tauri/src/sidecar_supervision.rs` (`cargo test sidecar_supervision`). + +## Meeting microphone manual checks + +1. Launch the desktop app, open a saved page, type `/meeting`, and select + **Meeting**. Click **Record** and accept the OS microphone permission prompt. + On macOS, check **System Settings → Privacy & Security → Microphone** if the + prompt was previously denied; reopen OpenBook after changing permission. +2. Speak, pause, and confirm a playable audio chunk appears. Resume, speak again, + then stop. Confirm both chunks play and the elapsed time excludes the pause. +3. With a supported transcription backend configured, confirm transcript lines + appear as chunks complete. If no backend is available, confirm audio and notes + remain usable and transcription offers retry. See + [local transcription setup](../../docs/local-transcription.md) for runtime + installation and the model download in **Settings → AI**. +4. Generate a summary, cancel a regeneration, and confirm the previous summary + survives. Type a manual note. Export audio and Markdown; check the ZIP's + individual recordings and the Markdown transcript, summary, and note. +5. Revoke microphone access and reopen the app. Attempt recording: confirm a + permission error appears, Stop stays disabled, and manual notes still work. + +These checks require the actual desktop webview and OS permission system. The +web e2e's oscillator stream does not validate desktop entitlements or hardware. diff --git a/packages/app/src-tauri/Info.plist b/packages/app/src-tauri/Info.plist new file mode 100644 index 00000000..4122baba --- /dev/null +++ b/packages/app/src-tauri/Info.plist @@ -0,0 +1,8 @@ + + + + + NSMicrophoneUsageDescription + OpenBook records meeting audio only when you press Record. + + diff --git a/packages/app/src-tauri/entitlements.plist b/packages/app/src-tauri/entitlements.plist index 48f7bf5c..8edce312 100644 --- a/packages/app/src-tauri/entitlements.plist +++ b/packages/app/src-tauri/entitlements.plist @@ -2,6 +2,8 @@ + com.apple.security.device.audio-input + com.apple.security.cs.allow-jit com.apple.security.cs.allow-unsigned-executable-memory diff --git a/packages/app/src-tauri/src/main.rs b/packages/app/src-tauri/src/main.rs index c3eb1acc..ad0e3568 100644 --- a/packages/app/src-tauri/src/main.rs +++ b/packages/app/src-tauri/src/main.rs @@ -15,6 +15,13 @@ //! dev` the host is unmanaged and the webview talks to the external `pnpm dev` //! server over loopback instead. Preferences (publish, token, book folder) //! persist in `host-config.json` under the app-data dir. +//! +//! Microphone permissions: the locked Wry 0.55.1 already implements +//! `WKUIDelegate::requestMediaCapturePermissionForOrigin` and grants the webview +//! layer's request. macOS still asks for and persists the user's OS permission; +//! Info.plist supplies its purpose string and entitlements.plist enables audio +//! input under the hardened runtime. Keep Wry's delegate (including its other +//! callbacks); replacing it would duplicate upstream behavior. See README.md. mod ipc; mod sidecar_supervision; diff --git a/packages/app/src/data/microphoneTransport.test.ts b/packages/app/src/data/microphoneTransport.test.ts new file mode 100644 index 00000000..0faa6aaf --- /dev/null +++ b/packages/app/src/data/microphoneTransport.test.ts @@ -0,0 +1,43 @@ +import {afterEach, expect, it, vi} from 'vitest'; +import {HttpDataClient, DEFAULT_MAX_ASSET_BYTES} from '@book.dev/sdk'; +import {invoke} from '@tauri-apps/api/core'; +import {tauriFetch} from './ipc'; + +vi.mock('@tauri-apps/api/core', () => ({invoke: vi.fn(), Channel: vi.fn()})); +vi.mock('@tauri-apps/api/event', () => ({listen: vi.fn()})); + +afterEach(() => vi.resetAllMocks()); + +it.each(['audio/webm', 'audio/mp4'])('round-trips two meeting-sized %s chunks through the desktop JSON transport', async (mime) => { + const stored = new Map(); + vi.mocked(invoke).mockImplementation(async (command, args) => { + expect(command).toBe('api_request'); + const request = args as {method: string; path: string; body: string}; + if (request.method === 'POST') { + expect(request.path).toContain('/api/assets?pageId=meeting'); + const body = JSON.parse(request.body) as {data: string; mime: string}; + expect(body.mime).toBe(mime); + expect(body.data.length).toBe(480000); + expect(new TextEncoder().encode(request.body).length).toBeLessThan(Math.ceil(DEFAULT_MAX_ASSET_BYTES * 4 / 3) + 64 * 1024); + const id = `chunk-${stored.size}`; + stored.set(id, body); + return {status: 201, headers: [], body: JSON.stringify({id})}; + } + const url = new URL(request.path, 'http://localhost'); + expect(url.searchParams.get('encoding')).toBe('base64'); + return {status: 200, headers: [], body: JSON.stringify(stored.get(url.pathname.split('/').pop()!))}; + }); + const client = new HttpDataClient('', undefined, {fetchImpl: tauriFetch}); + for (let chunk = 0; chunk < 2; chunk++) { + // 45 seconds at the recorder's 64 kbit/s target. Include every byte value + // so UTF-8 coercion or binary-body corruption cannot pass this check. + const bytes = Uint8Array.from({length: 360000}, (_, i) => (i + chunk) % 256); + const {id} = await client.putAsset(bytes, mime, 'meeting'); + const playback = await client.getAsset(id); + if (!playback) throw new Error('Uploaded audio was not returned for playback'); + expect(playback.mime).toBe(mime); + expect(playback.bytes.length).toBe(bytes.length); + expect(playback.bytes.every((byte, i) => byte === bytes[i])).toBe(true); + } + expect(stored.size).toBe(2); +}); diff --git a/packages/mcp/README.md b/packages/mcp/README.md index ce54205f..9d0002ff 100644 --- a/packages/mcp/README.md +++ b/packages/mcp/README.md @@ -93,6 +93,7 @@ Call `list_block_types` for the machine-readable source of truth: every entry ha | `tooltipcard` | `term:string`, `tip:string` | | `dbview` | `pageId:string` | | `dbform` | `databaseId:string`, `viewId:string` | +| `meeting` | `status:"idle"/"recording"/"processing"/"done"`, `audioChunks:[{assetId:string,durationMs:number,startedAtMs?:number}]`, `transcript:[{startMs:number,endMs:number,text:string}]`, `summary:string`, `startedAt:number`, `title:string`; container children are manual notes; [representation contract](../../docs/meeting-block.md) | | `form` | `formId:string`, `submissionKey:string`, `enabled:boolean`, `databaseId:string`, `schema:object`, `label:string`, `description:string` | Shared input/frame props are `name`, `label`, and `description` strings plus `compact` and `interactive` booleans. A structured option's `value` defaults to a slug of its `label`; `selected` contains those string values. diff --git a/packages/mcp/scripts/blockTypes.test.mts b/packages/mcp/scripts/blockTypes.test.mts index 85f78bba..230b5b1f 100644 --- a/packages/mcp/scripts/blockTypes.test.mts +++ b/packages/mcp/scripts/blockTypes.test.mts @@ -157,6 +157,22 @@ async function main(): Promise { check('an invalid structured option is refused with a typed error naming opts', isError(badOpts) && /"opts"/.test(resultText(badOpts)) && /object/i.test(resultText(badOpts))); + const meetingProps = {status: 'done', audioChunks: [{assetId: 'audio-1', durationMs: 500, startedAtMs: 0}], transcript: [{startMs: 0, endMs: 500, text: 'Hello'}], summary: 'Agreed', startedAt: 1234, title: 'Review'}; + const meetingPage = await seed.savePage({name: 'Meeting target', data: {editor: 'blocks', blockdoc: {blocks: [{id: 'meeting1', type: 'meeting', children: [{id: 'note1', type: 'paragraph', text: [{t: 'Notes'}]}]}]}, editorjs: {blocks: []}, values: [], names: []}}); + const meetingUpdate = await mcp.client.callTool({name: 'update_block_props', arguments: {pageId: meetingPage.id, blockId: 'meeting1', props: meetingProps}}); + check('meeting structured props update is accepted', !isError(meetingUpdate)); + const meetingCreate = await mcp.client.callTool({name: 'append_blocks', arguments: {pageId: meetingPage.id, blocks: [{type: 'meeting', props: meetingProps, children: [{type: 'paragraph', text: 'Manual notes'}]}]}}); + check('meeting creation with manual note children is accepted', !isError(meetingCreate)); + for (const props of [{status: 'paused'}, {audioChunks: [{assetId: 'a', durationMs: -1}]}, {transcript: [{startMs: 2, endMs: 1, text: 'Reversed'}]}]) { + const badCreate = await mcp.client.callTool({name: 'append_blocks', arguments: {pageId: meetingPage.id, blocks: [{type: 'meeting', props}]}}); + const badUpdate = await mcp.client.callTool({name: 'update_block_props', arguments: {pageId: meetingPage.id, blockId: 'meeting1', props}}); + check(`meeting rejects invalid creation and update: ${JSON.stringify(props)}`, isError(badCreate) && isError(badUpdate)); + } + const meetingListing = JSON.parse(resultText(await mcp.client.callTool({name: 'list_block_types', arguments: {types: ['meeting']}}))); + assert.equal(meetingListing.blocks.length, 1); + assert.deepEqual(meetingListing.blocks[0].propsSchema.properties.transcript.items.required, ['startMs', 'endMs', 'text']); + check('meeting listing publishes container nature and structured propsSchema', meetingListing.blocks[0].nature === 'container'); + console.log('\nAPI-2: plugin block types — rejected while uninstalled, accepted once installed'); const uninstalled = await mcp.client.callTool({ name: 'create_artifact_page', diff --git a/packages/mcp/src/server.ts b/packages/mcp/src/server.ts index f5fdb6a1..1942e7cd 100644 --- a/packages/mcp/src/server.ts +++ b/packages/mcp/src/server.ts @@ -25,6 +25,7 @@ import { findUnknownBlockType, FORM_FIELD_KINDS, invalidBlockProps, + invalidBlockTreeProps, insertBlocks, isHttpUrl, KIT_VALUE_BLOCK_TYPES, @@ -609,7 +610,7 @@ function nestedBlockSchema(typeDesc: string, propsDesc: string): z.ZodType - blockTreeError(blocks, {maxDepth: MAX_BLOCK_DEPTH, maxNodes: MAX_BLOCK_NODES}); + blockTreeError(blocks, {maxDepth: MAX_BLOCK_DEPTH, maxNodes: MAX_BLOCK_NODES}) ?? invalidBlockTreeProps(blocks); /** * The write-tool kind an MCP mutation maps to (the same identifiers the in-app diff --git a/packages/sdk/src/ai.ts b/packages/sdk/src/ai.ts index 5212ad79..1a0662e6 100644 --- a/packages/sdk/src/ai.ts +++ b/packages/sdk/src/ai.ts @@ -212,6 +212,8 @@ export interface AiUsageResponse { } export interface AiStatus { + /** Local audio capability is independent of the chat provider. */ + transcription?: {model: string; modelPresent: boolean; runtimeAvailable: boolean; ready: boolean; downloadUrl: string; detail?: string}; config: AiConfig; /** The engine can generate text right now. */ ready: boolean; diff --git a/packages/sdk/src/blockCatalogue.test.ts b/packages/sdk/src/blockCatalogue.test.ts index 0ef41ead..4310c0e1 100644 --- a/packages/sdk/src/blockCatalogue.test.ts +++ b/packages/sdk/src/blockCatalogue.test.ts @@ -17,6 +17,7 @@ import { CONTAINER_BLOCK_TYPES, findUnknownBlockType, invalidBlockProps, + invalidBlockTreeProps, isPluginBlockType, KNOWN_BLOCK_TYPE_IDS, MAX_BLOCK_DEPTH, @@ -242,3 +243,48 @@ describe('generated tool text', () => { for (const type of KNOWN_BLOCK_TYPE_IDS) expect(guidance).toContain(type); }); }); + + +describe('meeting representation contract (MEET-4)', () => { + it('is a kit container with typed structured props and no reactive value', () => { + expect(blockTypeInfo('meeting')).toMatchObject({category: 'kit', nature: 'container', kitValue: false}); + expect(blockTreeError([{type: 'meeting', children: [{type: 'paragraph', text: 'Manual notes'}]}])).toBeNull(); + for (const status of ['idle', 'recording', 'processing', 'done']) { + expect(invalidBlockProps('meeting', {status, startedAt: 1234, title: 'Review', summary: 'Agreed.', + audioChunks: [{assetId: 'audio-1', durationMs: 500, startedAtMs: 0}, {assetId: 'audio-2', durationMs: 250}], + transcript: [{startMs: 0, endMs: 500, text: 'Hello'}], + })).toBeNull(); + } + expect(invalidBlockProps('meeting', {})).toBeNull(); + expect(invalidBlockProps('meeting', {status: null, audioChunks: null, transcript: null, summary: null, startedAt: null, title: null})).toBeNull(); + }); + + it.each([ + {status: 'paused'}, {startedAt: -1}, {startedAt: '2026-01-01'}, {summary: []}, {title: 1}, + {audioChunks: ['asset']}, {audioChunks: [{assetId: 'a'}]}, {audioChunks: [{assetId: '', durationMs: 1}]}, + {audioChunks: [{assetId: 'a', durationMs: -1}]}, {audioChunks: [{assetId: 'a', durationMs: 1, startedAtMs: -1}]}, + {audioChunks: [{assetId: 'a', durationMs: Infinity}]}, {audioChunks: [{assetId: 'a', durationMs: 1, extra: true}]}, + {transcript: [{startMs: 0, text: 'Missing end'}]}, {transcript: [{startMs: 2, endMs: 1, text: 'Reversed'}]}, + {transcript: [{startMs: -1, endMs: 1, text: 'Negative'}]}, {transcript: [{startMs: 0, endMs: 1, text: 1}]}, + ])('rejects malformed props: %j', (props) => { + expect(invalidBlockProps('meeting', props)).toContain('Invalid prop'); + }); + + it('publishes nested item schemas and required fields to agents', () => { + const {blocks} = JSON.parse(blockCatalogueText([], ['meeting'])); + expect(blocks).toHaveLength(1); + expect(blocks[0].propsSchema.properties.status.enum).toEqual(['idle', 'recording', 'processing', 'done']); + expect(blocks[0].propsSchema.properties.audioChunks.items.required).toEqual(['assetId', 'durationMs']); + expect(blocks[0].propsSchema.properties.transcript.items.required).toEqual(['startMs', 'endMs', 'text']); + }); +}); + + +describe('creation prop validation', () => { + it('validates nested blocks with the same schemas as prop updates', () => { + expect(invalidBlockTreeProps([{type: 'group', children: [{type: 'meeting', props: {status: 'paused'}}]}])).toContain('Invalid prop "status"'); + expect(invalidBlockTreeProps([{type: 'heading', props: {level: 'two'}}])).toContain('Invalid prop "level"'); + expect(invalidBlockTreeProps([{type: 'meeting', props: []}])).toContain('expected an object'); + expect(invalidBlockTreeProps([{type: 'meeting'}, {type: 'plugin/widget', props: {anything: true}}, {type: 'meeting', props: {title: null, future: {x: 1}}}])).toBeNull(); + }); +}); diff --git a/packages/sdk/src/blockCatalogue.ts b/packages/sdk/src/blockCatalogue.ts index ae2dc957..2c86bf60 100644 --- a/packages/sdk/src/blockCatalogue.ts +++ b/packages/sdk/src/blockCatalogue.ts @@ -31,7 +31,7 @@ export {BLOCK_PROP_JSON_SCHEMAS, BLOCK_PROP_SCHEMAS} from './blockPropSchemas'; /** How a block stores content: `container` blocks carry child blocks in * `children`, `text` blocks carry rich text in `text`, `void` blocks carry - * only `props` (all kit widgets are void). */ + * only `props`. Kit blocks may also be containers. */ export type BlockNature = 'container' | 'text' | 'void'; /** The value shapes per-type prop validation understands. Deliberately coarse @@ -109,6 +109,7 @@ const CATALOGUE_LITERAL = [ {type: 'tooltipcard', category: 'kit', nature: 'void', props: {term: 'string', tip: 'string'}, hint: '{term,tip}'}, {type: 'dbview', category: 'kit', nature: 'void', props: {pageId: 'string'}, hint: 'embedded live database view {pageId} — the page hosting the database'}, {type: 'dbform', category: 'kit', nature: 'void', props: {databaseId: 'string', viewId: 'string'}, hint: 'embedded database form {databaseId,viewId} — a live reference, never a copied schema or capability'}, + {type: 'meeting', category: 'kit', nature: 'container', kitValue: false, props: {status: 'string', audioChunks: 'array', transcript: 'array', summary: 'string', startedAt: 'number', title: 'string'}, hint: 'meeting recording {status?,audioChunks?:[{assetId,durationMs,startedAtMs?}],transcript?:[{startMs,endMs,text}],summary?,startedAt?,title?}; times in ms, startedAt is Unix epoch; children hold manual notes; see docs/meeting-block.md'}, {type: 'form', category: 'kit', nature: 'void', kitValue: false, props: {formId: 'string', submissionKey: 'string', enabled: 'boolean', databaseId: 'string', schema: 'object', label: 'string', description: 'string'}, hint: 'public form definition {formId,submissionKey,enabled,databaseId?,schema}'}, ] as const satisfies readonly BlockTypeInfo[]; @@ -134,9 +135,9 @@ export const KIT_VALUE_BLOCK_TYPES: ReadonlySet = new Set( BLOCK_TYPE_CATALOGUE.filter((entry) => entry.kitValue === true).map((entry) => entry.type), ); -/** Core types whose `children` hold ordinary blocks. */ -export const CONTAINER_BLOCK_TYPES: ReadonlySet = new Set( - BLOCK_TYPE_CATALOGUE.filter((e) => e.nature === 'container').map((e) => e.type as CoreBlockType), +/** Catalogued types whose `children` hold ordinary blocks (including kit containers). */ +export const CONTAINER_BLOCK_TYPES: ReadonlySet = new Set( + BLOCK_TYPE_CATALOGUE.filter((e) => e.nature === 'container').map((e) => e.type), ); /** Core types that carry editable rich text. */ @@ -284,7 +285,7 @@ export function blockTreeError( structural = parentType ? `A "${type}" block must be a direct child of a "${needs}" block, not a "${parentType}" block (at "${here}").` : `A "${type}" block can't be top-level — it belongs directly inside a "${needs}" block (at "${here}").`; - } else if (hasChildren && !CONTAINER_BLOCK_TYPES.has(type as CoreBlockType)) { + } else if (hasChildren && !CONTAINER_BLOCK_TYPES.has(type)) { structural = `A "${type}" block can't hold children — only container blocks (${[...CONTAINER_BLOCK_TYPES].join(', ')}) do (at "${here}"). The nested blocks would be dropped.`; } else if (type === 'table' && hasChildren) { structural = raggedTableError(children, here); @@ -339,6 +340,30 @@ export function invalidBlockProps(type: string, props: Record): return `Invalid prop "${prop}" of a "${type}" block: ${issue.message} (got ${clipJson(props[prop])}).`; } +/** Validate declared props throughout a creation payload. Call blockTreeError + * first to enforce structure/size limits. Plugin and unknown props retain the + * same permissive behavior as update_block_props. */ +export function invalidBlockTreeProps(blocks: readonly unknown[], depth = 1): string | null { + if (depth > TYPE_WALK_MAX_DEPTH) return null; // structure validation owns depth errors + for (const raw of blocks) { + if (!raw || typeof raw !== 'object') continue; + const block = raw as {type?: unknown; props?: unknown; children?: unknown}; + const type = String(block.type ?? ''); + if (block.props !== undefined) { + if (!block.props || typeof block.props !== 'object' || Array.isArray(block.props)) { + return `Invalid props of a "${type}" block: expected an object.`; + } + const error = invalidBlockProps(type, block.props as Record); + if (error) return error; + } + if (Array.isArray(block.children)) { + const error = invalidBlockTreeProps(block.children, depth + 1); + if (error) return error; + } + } + return null; +} + const clipJson = (v: unknown): string => { const s = JSON.stringify(v) ?? String(v); return s.length > 40 ? `${s.slice(0, 40)}…` : s; @@ -388,10 +413,10 @@ export function addBlocksGuidance(): string { 'Append rich blocks to a page — text, layouts, tables, media, interactive inputs, and charts. User approves before they are added.', 'Each block is {type, text?, props?, children?}. `text` is a plain string (or rich runs [{"t","a":{b,i,u,s,c,a}}]); `children` nests blocks inside containers. Call list_block_types for the full catalogue including installed plugin blocks.', `TEXT: ${core.filter((e) => e.nature === 'text' && !e.parent).map(hinted).join('; ')}.`, - `CONTAINERS (use children): ${core.filter((e) => e.nature === 'container' || e.parent).map(hinted).join('; ')}. Give every table row the same number of cells.`, + `CONTAINERS (use children): ${BLOCK_TYPE_CATALOGUE.filter((e) => e.nature === 'container' || e.parent).map(hinted).join('; ')}. Give every table row the same number of cells.`, `MEDIA/OTHER: ${core.filter((e) => e.nature === 'void').map(hinted).join('; ')}.`, `INPUTS (each publishes props.name into the reactive scope): ${kit.filter((e) => e.kitValue).map(hinted).join('; ')}.`, - `REACTIVE DISPLAY/ACTIONS (props.source is a JS expression over input names): ${kit.filter((e) => !e.kitValue).map(hinted).join('; ')}.`, + `REACTIVE DISPLAY/ACTIONS (props.source is a JS expression over input names): ${kit.filter((e) => !e.kitValue && e.nature !== 'container').map(hinted).join('; ')}.`, 'Example: a budget widget → [{"type":"heading","text":"Budget","props":{"level":2}},{"type":"columns","children":[{"type":"column","props":{"span":5},"children":[{"type":"slider","props":{"name":"spent","label":"Spent","value":80,"min":0,"max":200}},{"type":"number","props":{"name":"budget","label":"Budget","value":120}}]},{"type":"column","props":{"span":7},"children":[{"type":"kitchart","props":{"kind":"bar","title":"Spent vs budget","labels":"Spent, Budget","source":"[spent, budget]"}},{"type":"statuslight","props":{"label":"On track","source":"budget - spent","okAt":0,"warnAt":-20}}]}]}].', ].join('\n'); } diff --git a/packages/sdk/src/blockPropSchemas.ts b/packages/sdk/src/blockPropSchemas.ts index 0daf2dcf..56e07684 100644 --- a/packages/sdk/src/blockPropSchemas.ts +++ b/packages/sdk/src/blockPropSchemas.ts @@ -43,6 +43,22 @@ const opts = array(option, 'Structured selectable options.'); const attrs = object({b: boolean(), i: boolean(), u: boolean(), s: boolean(), c: boolean(), a: string('Safe http, https, or mailto link.')}); const run = object({t: string('Run text.'), a: attrs}, 'One rich-text run.', ['t']); const runs = array(run, 'Rich-text runs.'); +// MEET-4 representation contract: docs/meeting-block.md. Nested objects are +// strict; top-level props retain the catalogue's nullable patch semantics. +const audioChunk = object({ + assetId: {schema: z.string().min(1).max(512), json: {type: 'string', minLength: 1, maxLength: 512}}, + durationMs: number('Chunk duration in milliseconds.', 0), + startedAtMs: number('Offset from the meeting start in milliseconds.', 0), +}, 'One audio asset in capture order.', ['assetId', 'durationMs']); +const transcriptSegment = object({ + startMs: number('Inclusive offset from the meeting start in milliseconds.', 0), + endMs: number('Exclusive offset from the meeting start in milliseconds; must be >= startMs.', 0), + text: string('Plain transcript text.'), +}, 'One transcript segment in display order.', ['startMs', 'endMs', 'text']); +transcriptSegment.schema = transcriptSegment.schema.refine( + (segment) => segment.endMs >= segment.startMs, + {message: 'endMs must be greater than or equal to startMs', path: ['endMs']}, +); const common = {bg: text}; const frame = {name: text, label: text, description: text, compact: boolean(), interactive: boolean()}; const inputText = {...frame, value: text, placeholder: text}; @@ -71,6 +87,14 @@ const fields = { formula: {...frame, source: expression('Expression evaluated over the reactive scope.')}, linkcard: {title: text, url: text, description: text}, tooltipcard: {term: text, tip: text}, dbview: {pageId: id}, dbform: {databaseId: id, viewId: id}, + meeting: { + status: enumeration(['idle', 'recording', 'processing', 'done'], 'Absent means idle.'), + audioChunks: array(audioChunk, 'Audio chunks in capture order; replace the entire array when updating.'), + transcript: array(transcriptSegment, 'Plain-text segments in display order; replace the entire array when updating.'), + summary: string('Plain-text summary.'), + startedAt: number('Meeting start as Unix epoch milliseconds.', 0), + title: text, + }, form: {formId: id, submissionKey: text, enabled: boolean(), databaseId: id, schema: freeObject, label: text, description: text}, } satisfies Record>; diff --git a/packages/sdk/src/client.ts b/packages/sdk/src/client.ts index 013fb178..b1b50375 100644 --- a/packages/sdk/src/client.ts +++ b/packages/sdk/src/client.ts @@ -209,6 +209,8 @@ export interface DataClient { aiSearch(query: string, limit?: number): Promise; aiTasks(goal: string, context?: string): Promise; aiDownloadModel(url?: string): Promise; + /** False for clients without a transcription transport (for example the local store). */ + readonly supportsTranscription?: boolean; transcribeAsset(assetId: string, pageId: string): Promise; aiComplete(text: string, onToken: (token: string) => void, opts?: {instruction?: string; signal?: AbortSignal}): Promise; aiGenerate(prompt: string, onToken: (token: string) => void, opts?: {system?: string; maxTokens?: number; signal?: AbortSignal}): Promise; diff --git a/packages/sdk/src/index.ts b/packages/sdk/src/index.ts index bae92bf9..c3f2d8af 100644 --- a/packages/sdk/src/index.ts +++ b/packages/sdk/src/index.ts @@ -563,6 +563,7 @@ export { unknownBlockTypeMessage, blockTreeError, invalidBlockProps, + invalidBlockTreeProps, blockCatalogueText, addBlocksGuidance, type BlockNature, diff --git a/packages/server/src/ai/agent.ts b/packages/server/src/ai/agent.ts index d63033e7..a3f57675 100644 --- a/packages/server/src/ai/agent.ts +++ b/packages/server/src/ai/agent.ts @@ -19,6 +19,7 @@ import { CONTAINER_BLOCK_TYPES, findUnknownBlockType, invalidBlockProps, + invalidBlockTreeProps, KIT_VALUE_BLOCK_TYPES, providerSettings, movePageTool, @@ -1060,6 +1061,8 @@ export class AgentRunner { if (structural) return structural; const bad = unknownBlockTypeMessage(findUnknownBlockType(blocks, {installedPluginIds: await this.installedPluginIds()})); if (bad) return bad; + const propError = invalidBlockTreeProps(blocks); + if (propError) return propError; const normalizeBlockText = (value: unknown): unknown => { if (!value || typeof value !== 'object' || Array.isArray(value)) return value; const block = value as Record; diff --git a/packages/server/src/ai/agentBlockTypes.test.ts b/packages/server/src/ai/agentBlockTypes.test.ts index eef33a35..c337fa50 100644 --- a/packages/server/src/ai/agentBlockTypes.test.ts +++ b/packages/server/src/ai/agentBlockTypes.test.ts @@ -261,3 +261,28 @@ describe('list_block_types', () => { expect(afterCatalogue.pluginBlocks.every((entry: {category: string}) => entry.category === 'plugin')).toBe(true); }); }); + + +describe('meeting agent surface (MEET-4)', () => { + it('accepts notes and structured props, rejects malformed creation and updates', async () => { + const page = await store.upsertPage({name: `meeting-${seq}`, data: { + editor: 'blocks', blockdoc: {blocks: [{id: 'meeting1', type: 'meeting'}]}, editorjs: {blocks: []}, values: [], names: [], + }}); + const props = {status: 'done', audioChunks: [{assetId: 'audio-1', durationMs: 500}], transcript: [{startMs: 0, endMs: 500, text: 'Hello'}]}; + const added = await runTool('add_blocks', {pageId: page.id, blocks: [{type: 'meeting', props, children: [{type: 'paragraph', text: 'Notes'}]}]}); + expect(added.result).toContain('SUGGESTED for review'); + expect((await runTool('update_block_props', {pageId: page.id, blockId: 'meeting1', props})).result).toContain('SUGGESTED for review'); + for (const invalid of [{status: 'paused'}, {audioChunks: [{assetId: 'a', durationMs: -1}]}, {transcript: [{startMs: 2, endMs: 1, text: 'Reversed'}]}]) { + for (const [tool, args] of [ + ['add_blocks', {pageId: page.id, blocks: [{type: 'meeting', props: invalid}]}], + ['update_block_props', {pageId: page.id, blockId: 'meeting1', props: invalid}], + ] as const) { + const response = await runTool(tool, args); + expect(response.result).toContain('Invalid prop'); + expect(response.events.some((event) => event.type === 'suggestions')).toBe(false); + } + } + const listed = JSON.parse((await runTool('list_block_types', {types: ['meeting']})).result); + expect(listed.blocks.find((block: {type: string}) => block.type === 'meeting').propsSchema.properties.transcript.items.required).toEqual(['startMs', 'endMs', 'text']); + }); +}); diff --git a/packages/server/src/ai/routes.ts b/packages/server/src/ai/routes.ts index 3cb003ef..36250df9 100644 --- a/packages/server/src/ai/routes.ts +++ b/packages/server/src/ai/routes.ts @@ -6,11 +6,16 @@ import type {PageStore} from '../store'; import type {AppEnv} from '../appEnv'; import {isLocalInstanceOwner, requireAuthenticatedRead, requireCreate, requireInstanceAdmin, requireInstanceOwner} from '../access'; import {AgentRunner, type AgentMessage} from './agent'; -import {TranscriptionConfigError, type AiService} from './service'; +import {ModelDownloadConfigError, TranscriptionConfigError, type AiService} from './service'; import {McpConfigError, type ExternalAgentTool, type McpClientManager} from './mcpClients'; +import {FixedWindowLimiter, clientIpKey} from '../agentTokens'; +import {LocalTranscriptionBusyError} from './whisper'; import type {TokenUsage} from './providers'; import type {AiUsageLog, UsageKind} from './usage'; +export const LOCAL_TRANSCRIPTION_RATE_LIMIT = 6; +const LOCAL_TRANSCRIPTION_RATE_WINDOW_MS = 60_000; + /** * The `/api/ai/*` surface. Generation endpoints stream tokens as SSE * (`data: {"token": "..."}` frames, closed by `data: {"done": true}`); @@ -22,6 +27,7 @@ import type {AiUsageLog, UsageKind} from './usage'; * logging failure never breaks the request. */ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore, onPagesChanged?: () => Promise, aiUsage?: AiUsageLog, mcp?: McpClientManager): void { + const localTranscriptionLimiter = new FixedWindowLimiter(LOCAL_TRANSCRIPTION_RATE_LIMIT, LOCAL_TRANSCRIPTION_RATE_WINDOW_MS); /** * Log a single generate/complete usage row against the effective provider/model. * Best-effort (the logger swallows its own errors); does nothing without a logger @@ -195,12 +201,20 @@ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore try { const {engine, provider, model} = await ai.transcriptionBackend(); await requirePaidInferenceAccess(c, store, provider === 'openai-compat' ? 'openai' : 'off'); + if (provider === 'local' && localTranscriptionLimiter.exceeded(clientIpKey(c))) { + c.header('Retry-After', String(LOCAL_TRANSCRIPTION_RATE_WINDOW_MS / 1000)); + return c.json({error: 'Too many local transcription requests. Please retry shortly.'}, 429); + } const result = await engine.transcribe(asset.bytes, {mime: asset.mime, signal: c.req.raw.signal}); // Audio backends do not report token counts. Keep unknown cloud cost null. await aiUsage?.log({provider, model, kind: 'transcribe', principal, usage: {inputTokens: 0, outputTokens: 0}}); return c.json(result); } catch (err) { if (err instanceof HTTPException) throw err; + if (err instanceof LocalTranscriptionBusyError) { + c.header('Retry-After', String(err.retryAfterSeconds)); + return c.json({error: err.message}, 429); + } if (err instanceof TranscriptionConfigError) return c.json({error: err.message}, 400); return c.json({error: 'Transcription failed. Check the provider in Settings → AI and retry.'}, 502); } @@ -211,7 +225,12 @@ export function mountAiRoutes(app: Hono, ai: AiService, store: PageStore // surface) — only the trusted instance owner may supply it. await requireInstanceOwner(c, store); const {url} = (await c.req.json().catch(() => ({}))) as {url?: string}; - return c.json(await ai.startDownload(url)); + try { + return c.json(await ai.startDownload(url)); + } catch (err) { + if (err instanceof ModelDownloadConfigError) return c.json({error: err.message}, 400); + throw err; + } }); // The agent harness: runs the tool loop against the library and streams diff --git a/packages/server/src/ai/service.ts b/packages/server/src/ai/service.ts index f03187b1..07aaf66a 100644 --- a/packages/server/src/ai/service.ts +++ b/packages/server/src/ai/service.ts @@ -5,6 +5,7 @@ import {providerSettings, type AiConfig, type AiProvider, type AiProviderSetting import type {Db} from '../db'; import {createEngine, MockEngine, OpenAiCompatEngine, type TranscriptionEngine, type AiEngine, type GenerateOptions} from './providers'; import {assembleSearchResults, bm25Scores, buildIndex, cosine, pageRowsToDocs, parseTaskList, type Bm25Index} from './search'; +import {WHISPER_MODEL, WHISPER_MODEL_URL} from './whisper'; import {SkillStore} from './skills'; /** @@ -96,6 +97,7 @@ interface DownloadState { } export class TranscriptionConfigError extends Error {} +export class ModelDownloadConfigError extends Error {} export class AiService { private config: AiConfig = DEFAULT_CONFIG; @@ -114,6 +116,7 @@ export class AiService { private readonly modelsDir: string, /** MEET-3: lazily resolve the managed local audio backend. */ private readonly localTranscription?: () => Promise, + private readonly localLifecycle?: {status(): Promise>; dispose(): Promise}, ) { this.skills = new SkillStore(db); } @@ -162,9 +165,9 @@ export class AiService { return {engine: new OpenAiCompatEngine(audio.baseUrl?.trim() || 'https://api.openai.com', audio.model?.trim() || 'whisper-1', 'openai', audio.apiKey), provider: 'openai-compat', model: audio.model?.trim() || 'whisper-1'}; } const local = await this.localTranscription?.(); - if (local) return {engine: local, provider: 'local', model: audio?.model ?? 'local'}; + if (local) return {engine: local, provider: 'local', model: WHISPER_MODEL}; if (config.provider === 'mock') return {engine: new MockEngine(), provider: 'mock', model: 'mock'}; - throw new TranscriptionConfigError('Local transcription is unavailable. Configure transcription in Settings → AI.'); + throw new TranscriptionConfigError('Local transcription is unavailable. Download the Whisper model and check the local runtime in Settings → AI.'); } async status(): Promise { @@ -183,6 +186,7 @@ export class AiService { } return { config: this.config, + transcription: await this.localLifecycle?.status(), ready, embeddings, detail, @@ -336,7 +340,13 @@ export class AiService { async startDownload(url = DEFAULT_MODEL_URL): Promise { await this.loadConfig(); if (this.download && !this.download.done && !this.download.error) return this.download; - const fileName = decodeURIComponent(new URL(url).pathname.split('/').pop() || 'model.gguf'); + let fileName: string; + try { + fileName = decodeURIComponent(new URL(url).pathname.split('/').pop() || 'model.gguf'); + } catch { + throw new ModelDownloadConfigError('Invalid model download URL or filename encoding.'); + } + if (fileName !== path.basename(fileName) || fileName === '.' || fileName === '..' || fileName.includes('\\')) throw new ModelDownloadConfigError('Invalid model filename'); mkdirSync(this.modelsDir, {recursive: true}); const dest = path.join(this.modelsDir, fileName); const state: DownloadState = {url, received: 0, total: null, done: false}; @@ -367,7 +377,7 @@ export class AiService { state.done = true; } // Auto-select the downloaded model for the llama provider. - if (this.config.provider === 'llama' && !this.config.model) { + if (url !== WHISPER_MODEL_URL && this.config.provider === 'llama' && !this.config.model) { await this.setConfig({...this.config, model: fileName}); } } catch (err) { @@ -381,5 +391,6 @@ export class AiService { async dispose(): Promise { await this.engine?.dispose().catch(() => undefined); + await this.localLifecycle?.dispose(); } } diff --git a/packages/server/src/ai/whisper.test.ts b/packages/server/src/ai/whisper.test.ts new file mode 100644 index 00000000..a9adece0 --- /dev/null +++ b/packages/server/src/ai/whisper.test.ts @@ -0,0 +1,104 @@ +import {mkdtemp, readFile, readdir, rm, writeFile} from 'node:fs/promises'; +import {tmpdir} from 'node:os'; +import path from 'node:path'; +import {afterEach, beforeEach, describe, expect, it} from 'vitest'; +import {LocalWhisper, LocalTranscriptionBusyError, parseWhisperOutput, runWhisperProcess, WHISPER_MODEL} from './whisper'; + +let dir: string; +beforeEach(async () => { dir = await mkdtemp(path.join(tmpdir(), 'meet3-test-')); }); +afterEach(async () => { await rm(dir, {recursive: true, force: true}); }); + +async function script(name: string, source: string): Promise { + const file = path.join(dir, name); + await writeFile(file, `#!${process.execPath}\n${source}`, {mode: 0o700}); + return file; +} + +describe('optional local whisper runtime', () => { + it('returns null for missing model or executables, and discovers a model without restart', async () => { + const local = new LocalWhisper(dir, process.execPath, process.execPath); + expect(await local.resolve()).toBeNull(); + expect(await local.status()).toMatchObject({modelPresent: false, runtimeAvailable: true, ready: false}); + await writeFile(path.join(dir, WHISPER_MODEL), 'test model'); + expect(await local.resolve()).toBe(local); + const missing = new LocalWhisper(dir, path.join(dir, 'missing'), process.execPath); + expect(await missing.resolve()).toBeNull(); + expect(await missing.status()).toMatchObject({modelPresent: true, runtimeAvailable: false, detail: expect.stringContaining('Settings → AI')}); + await local.dispose(); + expect(await local.resolve()).toBeNull(); + }); + + it('decodes exact input bytes, converts millisecond offsets, and cleans scratch files', async () => { + const marker = path.join(dir, 'scratch'); + const ffmpeg = await script('ffmpeg', 'const fs = require(\'node:fs\'); const args = process.argv.slice(2); const input = args[args.indexOf(\'-i\') + 1]; if (fs.readFileSync(input).toString() !== \'recording\') process.exit(2); fs.writeFileSync(args.at(-1), \'wav\');'); + const whisper = await script('whisper', `const fs = require('node:fs'); const args = process.argv.slice(2); const output = args[args.indexOf('-of') + 1]; fs.writeFileSync(${JSON.stringify(marker)}, require('node:path').dirname(output)); fs.writeFileSync(output + '.json', JSON.stringify({transcription: [{offsets: {from: 250, to: 1500}, text: ' Hello world.'}]}));`); + const local = new LocalWhisper(dir, whisper, ffmpeg); + expect(await local.transcribe(new TextEncoder().encode('recording'), {filename: '../../escape.webm', mime: 'audio/webm'})) + .toEqual({text: 'Hello world.', segments: [{start: 0.25, end: 1.5, text: 'Hello world.'}], durationMs: 1500}); + await expect(readdir(await readFile(marker, 'utf8'))).rejects.toThrow(); + }); + + it.each(['abort', 'dispose'] as const)('%s kills in-flight inference and removes scratch files', async (action) => { + const marker = path.join(dir, 'started'); + const ffmpeg = await script('ffmpeg', 'process.exit(0);'); + const whisper = await script('whisper', `const fs = require('node:fs'); const args = process.argv.slice(2); fs.writeFileSync(${JSON.stringify(marker)}, JSON.stringify({pid: process.pid, dir: require('node:path').dirname(args[args.indexOf('-of') + 1])})); setInterval(() => {}, 1000);`); + const local = new LocalWhisper(dir, whisper, ffmpeg); + const controller = new AbortController(); + const promise = local.transcribe(new Uint8Array([1]), {signal: controller.signal}); + const rejected = expect(promise).rejects.toMatchObject({name: 'AbortError'}); + await expect.poll(async () => readFile(marker, 'utf8').catch(() => '')).not.toBe(''); + const started = JSON.parse(await readFile(marker, 'utf8')) as {pid: number; dir: string}; + if (action === 'abort') controller.abort(); + else await local.dispose(); + await rejected; + expect(() => process.kill(started.pid, 0)).toThrow(); + await expect(readdir(started.dir)).rejects.toThrow(); + }); + + it.each(['abort', 'failure'] as const)('limits jobs to two and releases permits after %s', async (outcome) => { + const mode = path.join(dir, 'mode'); + await writeFile(mode, 'wait'); + const ffmpeg = await script('ffmpeg', 'process.exit(0);'); + const whisper = await script('whisper', `const fs = require('node:fs'); const args = process.argv.slice(2); const output = args[args.indexOf('-of') + 1]; fs.writeFileSync(${JSON.stringify(dir)} + '/' + process.pid + '.started', ''); setInterval(() => { const mode = fs.readFileSync(${JSON.stringify(mode)}, 'utf8'); if (mode === 'fail') process.exit(1); if (mode === 'ok') { fs.writeFileSync(output + '.json', JSON.stringify({transcription: []})); process.exit(0); } }, 10);`); + const local = new LocalWhisper(dir, whisper, ffmpeg); + const controller = new AbortController(); + const jobs = Promise.allSettled([0, 1].map(() => local.transcribe(new Uint8Array([1]), {signal: controller.signal}))); + try { + await expect.poll(async () => (await readdir(dir)).filter((f) => f.endsWith('.started')).length).toBe(2); + await expect(local.transcribe(new Uint8Array([1]))).rejects.toBeInstanceOf(LocalTranscriptionBusyError); + if (outcome === 'abort') controller.abort(); + else await writeFile(mode, 'fail'); + const results = await jobs; + for (const result of results) { + expect(result.status).toBe('rejected'); + if (result.status === 'rejected') expect(result.reason.message).toContain(outcome === 'abort' ? 'abort' : 'Local audio processing failed'); + } + await writeFile(mode, 'ok'); + const recovered = await Promise.all([0, 1].map(() => local.transcribe(new Uint8Array([1])))); + expect(recovered).toEqual([{text: '', segments: [], durationMs: 0}, {text: '', segments: [], durationMs: 0}]); + } finally { + await local.dispose(); + await jobs; + } + }); + + it.each([1001, 1000.6])('keeps duration integer milliseconds without a seconds round-trip (%s)', (end) => { + expect(parseWhisperOutput({transcription: [ + {offsets: {from: 0, to: end}, text: 'First'}, + {offsets: {from: 0, to: 900}, text: 'Second'}, + ]}).durationMs).toBe(1001); + }); + + it('rejects pre-aborted work, spawn failures, failed processes and malformed output', async () => { + const controller = new AbortController(); + controller.abort(); + await expect(new LocalWhisper(dir).transcribe(new Uint8Array(), {signal: controller.signal})).rejects.toMatchObject({name: 'AbortError'}); + const signal = new AbortController().signal; + await expect(runWhisperProcess(path.join(dir, 'missing'), [], signal)).rejects.toThrow(); + await expect(runWhisperProcess(process.execPath, ['-e', 'process.exit(1)'], signal)).rejects.toThrow('Local audio processing failed'); + for (const value of [null, {}, {transcription: [{text: 'bad', offsets: {from: -1, to: 100}}]}]) { + expect(() => parseWhisperOutput(value)).toThrow(); + } + expect(parseWhisperOutput({transcription: []})).toEqual({text: '', segments: [], durationMs: 0}); + }); +}); diff --git a/packages/server/src/ai/whisper.ts b/packages/server/src/ai/whisper.ts new file mode 100644 index 00000000..989b0858 --- /dev/null +++ b/packages/server/src/ai/whisper.ts @@ -0,0 +1,146 @@ +import {spawn} from 'node:child_process'; +import {constants} from 'node:fs'; +import {access, mkdtemp, readFile, rm, stat, writeFile} from 'node:fs/promises'; +import {tmpdir} from 'node:os'; +import path from 'node:path'; +import type {AiStatus, AiTranscriptionResult} from '@book.dev/sdk'; +import type {TranscriptionEngine, TranscribeOptions} from './providers'; + +export const LOCAL_TRANSCRIPTION_MAX_JOBS = 2; + +export class LocalTranscriptionBusyError extends Error { + readonly retryAfterSeconds = 5; + constructor() { + super('Local transcription is busy. Please retry shortly.'); + } +} + +export const WHISPER_MODEL = 'ggml-base.bin'; +export const WHISPER_MODEL_URL = 'https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.bin'; + +async function executable(command: string): Promise { + const candidates = path.isAbsolute(command) || command.includes(path.sep) + ? [command] : (process.env.PATH ?? '').split(path.delimiter).map((dir) => path.join(dir, command)); + for (const candidate of candidates) { + try { + await access(candidate, constants.X_OK); + if ((await stat(candidate)).isFile()) return candidate; + } catch { /* Optional system dependency. */ } + } + return null; +} + +/** One child per operation: no listening port or native Node ABI dependency. + * SIGKILL makes cancellation deterministic even inside a native inference call. + * Wait for close before removing its private scratch directory. */ +export async function runWhisperProcess(command: string, args: string[], signal: AbortSignal): Promise { + signal.throwIfAborted(); + await new Promise((resolve, reject) => { + const child = spawn(command, args, {stdio: 'ignore', shell: false}); + const abort = () => { child.kill('SIGKILL'); }; + signal.addEventListener('abort', abort, {once: true}); + if (signal.aborted) abort(); + child.once('error', (error) => { + signal.removeEventListener('abort', abort); + reject(error); + }); + child.once('close', (code) => { + signal.removeEventListener('abort', abort); + if (signal.aborted) reject(signal.reason); + else if (code !== 0) reject(new Error('Local audio processing failed. Check the recording and Settings → AI.')); + else resolve(); + }); + }); +} + +export function parseWhisperOutput(value: unknown): AiTranscriptionResult { + const data = value as {transcription?: {offsets?: {from?: number; to?: number}; text?: string}[]}; + if (!data || !Array.isArray(data.transcription)) throw new Error('Invalid whisper output'); + let durationMs = 0; + const segments = data.transcription.map((segment) => { + const start = segment?.offsets?.from; + const end = segment?.offsets?.to; + if (typeof start !== 'number' || !Number.isFinite(start) || start < 0 + || typeof end !== 'number' || !Number.isFinite(end) || end < start || typeof segment.text !== 'string') { + throw new Error('Invalid whisper segment'); + } + durationMs = Math.max(durationMs, end); + return {start: start / 1000, end: end / 1000, text: segment.text.trim()}; + }); + return {text: segments.map((s) => s.text).join(' ').trim(), segments, durationMs: Math.round(durationMs)}; +} + +/** Optional whisper.cpp + FFmpeg runtime. Models live beside chat models, but + * inference runs only on demand, releasing model memory after each recording. */ +export class LocalWhisper implements TranscriptionEngine { + private readonly active = new Set(); + private readonly pending = new Set>(); + private disposed = false; + + constructor( + private readonly modelsDir: string, + private readonly whisperCommand = process.env.OPENBOOK_WHISPER_BIN || 'whisper-cli', + private readonly ffmpegCommand = process.env.OPENBOOK_FFMPEG_BIN || 'ffmpeg', + ) {} + + async status(): Promise> { + const [whisper, ffmpeg, modelPresent] = await Promise.all([ + executable(this.whisperCommand), executable(this.ffmpegCommand), + stat(path.join(this.modelsDir, WHISPER_MODEL)).then((s) => s.isFile() && s.size > 0).catch(() => false), + ]); + const runtimeAvailable = Boolean(whisper && ffmpeg); + return { + model: WHISPER_MODEL, modelPresent, runtimeAvailable, ready: modelPresent && runtimeAvailable && !this.disposed, + downloadUrl: WHISPER_MODEL_URL, + detail: !runtimeAvailable ? 'Install whisper.cpp (whisper-cli) and FFmpeg on the server, then return to Settings → AI.' + : !modelPresent ? 'Download Whisper base in Settings → AI to enable local transcription.' : undefined, + }; + } + + async resolve(): Promise { + return (await this.status()).ready ? this : null; + } + + transcribe(bytes: Uint8Array, opts: TranscribeOptions = {}): Promise { + if (opts.signal?.aborted) return Promise.reject(opts.signal.reason); + // This set is a non-queuing semaphore shared by all requests to this server. + // Reserve synchronously, before any await or temporary-file/process creation; + // keep the permit until perform's finally has removed the scratch directory. + if (this.active.size >= LOCAL_TRANSCRIPTION_MAX_JOBS) return Promise.reject(new LocalTranscriptionBusyError()); + const controller = new AbortController(); + const signal = opts.signal ? AbortSignal.any([opts.signal, controller.signal]) : controller.signal; + this.active.add(controller); + const operation = this.perform(bytes, signal).finally(() => { + this.active.delete(controller); + this.pending.delete(operation); + }); + this.pending.add(operation); + return operation; + } + + private async perform(bytes: Uint8Array, signal: AbortSignal): Promise { + signal.throwIfAborted(); + if (this.disposed) throw new Error('Local transcription has stopped.'); + const dir = await mkdtemp(path.join(tmpdir(), 'openbook-whisper-')); + try { + const input = path.join(dir, 'input'); + const wav = path.join(dir, 'audio.wav'); + const output = path.join(dir, 'result'); + await writeFile(input, bytes, {signal}); + // Probe by content, never by an untrusted filename; accept browser WebM, + // MP4, Ogg and WAV alike. Restrict nested input protocols to local files. + await runWhisperProcess(this.ffmpegCommand, ['-nostdin', '-v', 'error', '-protocol_whitelist', 'file,pipe', '-i', input, '-vn', '-ar', '16000', '-ac', '1', '-c:a', 'pcm_s16le', wav], signal); + await runWhisperProcess(this.whisperCommand, ['-m', path.join(this.modelsDir, WHISPER_MODEL), '-f', wav, '-l', 'auto', '-oj', '-of', output], signal); + signal.throwIfAborted(); + return parseWhisperOutput(JSON.parse(await readFile(`${output}.json`, 'utf8'))); + } finally { + await rm(dir, {recursive: true, force: true}); + } + } + + async dispose(): Promise { + this.disposed = true; + for (const controller of this.active) controller.abort(); + await Promise.allSettled(this.pending); + } +} diff --git a/packages/server/src/localClient.ts b/packages/server/src/localClient.ts index 69001f19..11a37682 100644 --- a/packages/server/src/localClient.ts +++ b/packages/server/src/localClient.ts @@ -921,6 +921,8 @@ export class LocalDataClient implements DataClient { return Promise.reject(this.aiUnavailable()); } + readonly supportsTranscription = false; + transcribeAsset(): Promise { return Promise.reject(this.aiUnavailable()); } diff --git a/packages/server/src/server.ts b/packages/server/src/server.ts index bfbff2ba..a236505e 100644 --- a/packages/server/src/server.ts +++ b/packages/server/src/server.ts @@ -6,6 +6,7 @@ import {PageStore, PAGE_VERSION_KEEP, PAGE_VERSION_MAX_AGE_MS} from './store'; import {PageHub} from './hub'; import {BookMirror, MirrorLockedError, WriteBudgetError} from './mirror'; import {AiService} from './ai/service'; +import {LocalWhisper} from './ai/whisper'; import {McpClientManager} from './ai/mcpClients'; import {AiUsageLog} from './ai/usage'; import {IdentityService} from './instanceConfig'; @@ -436,7 +437,8 @@ export async function startServer(opts: StartOptions): Promise { // (server mode). The subsystem is inert until configured via /api/ai. const modelsDir = process.env.OPENBOOK_MODELS_DIR || (opts.dataDir ? path.join(opts.dataDir, 'models') : path.join(os.homedir(), '.openbook', 'models')); - const ai = new AiService(db, modelsDir); + const whisper = new LocalWhisper(modelsDir); + const ai = new AiService(db, modelsDir, () => whisper.resolve(), whisper); // External-tools (MCP client) manager (AGENT-3): owned beside AiService, pools // connections to admin-registered MCP servers and hands the agent route // namespaced `mcp__*` tools. Inert until an admin configures + enables a server diff --git a/packages/server/src/transcription.test.ts b/packages/server/src/transcription.test.ts index 6d6fce60..1f5b5a92 100644 --- a/packages/server/src/transcription.test.ts +++ b/packages/server/src/transcription.test.ts @@ -11,6 +11,9 @@ import {AiService} from './ai/service'; import {MockEngine, OpenAiCompatEngine} from './ai/providers'; import {AiUsageLog} from './ai/usage'; import {LocalDataClient} from './localClient'; +import {LocalWhisper, WHISPER_MODEL, WHISPER_MODEL_URL} from './ai/whisper'; +import {LOCAL_TRANSCRIPTION_RATE_LIMIT} from './ai/routes'; +import {readFile, readdir, writeFile} from 'node:fs/promises'; let db: PgliteDb; let store: PageStore; @@ -119,6 +122,134 @@ describe('transcription contract', () => { expect(local).toHaveBeenCalledOnce(); }); + it('uses the managed resolver fallback and exposes actionable local state through aiStatus', async () => { + const local = new LocalWhisper(dir, join(dir, 'missing'), join(dir, 'missing-ffmpeg')); + const service = new AiService(db, dir, () => local.resolve(), local); + const app = appWith(service); + expect((await post(app)).status).toBe(400); + expect((await (await post(app)).json()).error).toContain('Settings → AI'); + const status = await app.request(API.aiStatus, {headers: {...headers, [LOCAL_OWNER_HEADER]: secret}}); + expect((await status.json()).transcription).toMatchObject({modelPresent: false, runtimeAvailable: false, ready: false, downloadUrl: WHISPER_MODEL_URL}); + await service.setConfig({provider: 'mock'}); + expect((await post(app)).status).toBe(200); + await service.dispose(); + }); + + it('logs local transcription as free and downloads its model without selecting it for chat', async () => { + const service = new AiService(db, dir, async () => new MockEngine()); + const usage = new AiUsageLog(store); + expect((await post(appWith(service, usage))).status).toBe(200); + expect((await usage.report()).rows?.[0]).toMatchObject({provider: 'local', model: WHISPER_MODEL, kind: 'transcribe', cost: 0}); + await service.setConfig({provider: 'llama'}); + vi.stubGlobal('fetch', vi.fn(async () => new Response('model bytes'))); + const app = appWith(service); + const download = await app.request(API.aiModelDownload, {method: 'POST', headers: {...headers, [LOCAL_OWNER_HEADER]: secret}, body: JSON.stringify({url: WHISPER_MODEL_URL})}); + expect(download.status).toBe(200); + await expect.poll(async () => (await service.status()).download?.done).toBe(true); + expect(await readFile(join(dir, WHISPER_MODEL), 'utf8')).toBe('model bytes'); + expect((await service.getConfig()).model).toBeUndefined(); + expect(await readdir(dir)).not.toContain(`${WHISPER_MODEL}.part`); + await service.dispose(); + }); + + it('caps parallel local HTTP jobs across anonymous readers at two, returning 429 with Retry-After', async () => { + await store.updateInstanceConfig({ownerSubject: 'test#owner', guestAccess: 'read'}); + await store.setPageVisibility(pageId, 'public'); + const ffmpeg = join(dir, 'ffmpeg'); + const whisper = join(dir, 'whisper'); + const release = join(dir, 'release'); + await writeFile(ffmpeg, `#!${process.execPath}\nprocess.exit(0);`, {mode: 0o700}); + await writeFile(whisper, `#!${process.execPath}\nconst fs = require('node:fs'); const args = process.argv.slice(2); const output = args[args.indexOf('-of') + 1]; fs.writeFileSync(${JSON.stringify(dir)} + '/' + process.pid + '.started', ''); setInterval(() => { if (fs.existsSync(${JSON.stringify(release)})) { fs.writeFileSync(output + '.json', JSON.stringify({transcription: [{offsets: {from: 0, to: 1001}, text: 'Local'}]})); process.exit(0); } }, 10);`, {mode: 0o700}); + await writeFile(join(dir, WHISPER_MODEL), 'test model'); + const local = new LocalWhisper(dir, whisper, ffmpeg); + const service = new AiService(db, dir, () => local.resolve(), local); + const app = appWith(service); + const request = (ip: string) => app.request(API.aiTranscribe, { + method: 'POST', headers: {...headers, [FORWARDED_HEADER]: '1'}, body: JSON.stringify({assetId, pageId}), + }, {incoming: {socket: {remoteAddress: ip}}}); + const accepted = [request('192.0.2.1'), request('192.0.2.2')]; + try { + await expect.poll(async () => (await readdir(dir)).filter((f) => f.endsWith('.started')).length).toBe(2); + const busy = await request('192.0.2.3'); + expect(busy.status).toBe(429); + expect(busy.headers.get('Retry-After')).toBe('5'); + await writeFile(release, 'go'); + for (const response of await Promise.all(accepted)) { + expect(response.status).toBe(200); + expect(await response.json()).toMatchObject({text: 'Local', durationMs: 1001}); + } + expect((await request('192.0.2.3')).status).toBe(200); + } finally { + await service.dispose(); + await Promise.all(accepted); + } + }); + + it('rate-limits local requests per socket IP, ignores spoofed forwarding headers, and leaves mock/cloud untouched', async () => { + const engine = new MockEngine(); + const transcribe = vi.spyOn(engine, 'transcribe'); + let available = true; + const service = new AiService(db, dir, async () => available ? engine : null); + const app = appWith(service); + const request = (ip = '192.0.2.1', forwardedFor = '') => app.request(API.aiTranscribe, { + method: 'POST', headers: {...headers, [LOCAL_OWNER_HEADER]: secret, 'x-forwarded-for': forwardedFor}, body: JSON.stringify({assetId, pageId}), + }, {incoming: {socket: {remoteAddress: ip}}}); + for (let i = 0; i < LOCAL_TRANSCRIPTION_RATE_LIMIT; i++) expect((await request()).status).toBe(200); + const denied = await request('192.0.2.1', '198.51.100.1'); + expect(denied.status).toBe(429); + expect(denied.headers.get('Retry-After')).toBe('60'); + expect(transcribe).toHaveBeenCalledTimes(LOCAL_TRANSCRIPTION_RATE_LIMIT); + expect((await request('192.0.2.2')).status).toBe(200); + available = false; + await service.setConfig({provider: 'mock'}); + expect((await request()).status).toBe(200); + await service.setConfig({provider: 'off', transcription: {provider: 'openai-compat'}}); + vi.stubGlobal('fetch', vi.fn(async () => new Response(JSON.stringify({text: 'Cloud'})))); + expect((await request()).status).toBe(200); + available = true; + await service.setConfig({provider: 'off', transcription: {provider: 'local'}}); + vi.spyOn(Date, 'now').mockReturnValue(Date.now() + 60_001); + expect((await request()).status).toBe(200); + await service.dispose(); + }); + + it('returns 400 for malformed percent-encoding in model download filenames without fetching', async () => { + const fetchSpy = vi.spyOn(globalThis, 'fetch'); + const app = appWith(); + for (const filename of ['bad%.bin', 'bad%ZZ.bin', 'bad%E0%A4%A.bin']) { + const response = await app.request(API.aiModelDownload, { + method: 'POST', headers: {...headers, [LOCAL_OWNER_HEADER]: secret}, + body: JSON.stringify({url: `https://example.test/${filename}`}), + }); + expect(response.status).toBe(400); + expect(await response.json()).toEqual({error: 'Invalid model download URL or filename encoding.'}); + } + expect(fetchSpy).not.toHaveBeenCalled(); + expect((await ai.status()).download).toBeUndefined(); + }); + + // Opt-in: OPENBOOK_TEST_WHISPER=1, OPENBOOK_MODELS_DIR containing ggml-base.bin, + // whisper-cli + ffmpeg on PATH (or OPENBOOK_WHISPER_BIN / OPENBOOK_FFMPEG_BIN). + it.skipIf(process.env.OPENBOOK_TEST_WHISPER !== '1')('native whisper: POST transcribes synthesized WAV with no cloud keys', async () => { + const local = new LocalWhisper(process.env.OPENBOOK_MODELS_DIR || join(dir, 'models')); + expect((await local.status()).ready).toBe(true); + const wav = Buffer.alloc(44 + 16000 * 2); + wav.write('RIFF'); wav.writeUInt32LE(wav.length - 8, 4); wav.write('WAVEfmt ', 8); + wav.writeUInt32LE(16, 16); wav.writeUInt16LE(1, 20); wav.writeUInt16LE(1, 22); + wav.writeUInt32LE(16000, 24); wav.writeUInt32LE(32000, 28); wav.writeUInt16LE(2, 32); wav.writeUInt16LE(16, 34); + wav.write('data', 36); wav.writeUInt32LE(wav.length - 44, 40); + assetId = (await store.putAsset(wav, 'audio/wav')).id; + await store.refAsset(assetId, pageId); + const service = new AiService(db, dir, () => local.resolve(), local); + try { + const response = await post(appWith(service)); + expect(response.status).toBe(200); + expect(await response.json()).toMatchObject({text: expect.any(String), segments: expect.any(Array), durationMs: expect.any(Number)}); + } finally { + await service.dispose(); + } + }, 120_000); + it('returns identical 404s for missing, unreadable, unreferenced, and unrelated assets/pages', async () => { await ai.setConfig({provider: 'mock'}); await store.updateInstanceConfig({ownerSubject: 'test#owner', guestAccess: 'read'}); diff --git a/packages/ui/src/blockeditor/BlockEditor.tsx b/packages/ui/src/blockeditor/BlockEditor.tsx index cb8f9675..30e1539a 100644 --- a/packages/ui/src/blockeditor/BlockEditor.tsx +++ b/packages/ui/src/blockeditor/BlockEditor.tsx @@ -2584,7 +2584,10 @@ const BlockBody: React.FC = ({block, ...shared}) stays live for a reader; `pageReadOnly` is the document's real lock, which that override must not hide from a block that offers to write somewhere else. See {@link CustomBlockProps.pageReadOnly}. */} - + + {custom.type === 'meeting' && blockChildren(block) && + } + ); } diff --git a/packages/ui/src/blockeditor/MeetingBlockView.tsx b/packages/ui/src/blockeditor/MeetingBlockView.tsx new file mode 100644 index 00000000..453285a6 --- /dev/null +++ b/packages/ui/src/blockeditor/MeetingBlockView.tsx @@ -0,0 +1,124 @@ +import {useEffect, useMemo, useState, useSyncExternalStore} from 'react'; +import {t} from '@/i18n'; +import {assetBridge, subscribeAssetBridge} from '@/lib/assetBridge'; +import {getPageIdForDoc} from '@/lib/aiBridge'; +import {blockChildren, blockProp, insertBlock} from './model'; +import type {CustomBlockDef, CustomBlockProps} from './registry'; +import {exportMeetingAudio} from './meetingAudioExport'; +import {MeetingSummary} from './MeetingSummary'; +import {useKitLock} from './kit/lock'; +import {KitFrame, NameDescriptionFields} from './kit/KitFrame'; +import {chunkKey, meetingRecorder, meetingTime, type MeetingAudio, type MeetingSegment} from './meetingRecorder'; + +function ChunkAudio({audio, blob}: {audio: MeetingAudio; blob?: Blob}) { + const [url, setUrl] = useState(''); + useEffect(() => { + let disposed = false; + let objectUrl = ''; + const load = async (): Promise => { + const asset = blob ? null : await assetBridge.getAsset(audio.assetId); + const data = blob ?? (asset ? new Blob([new Uint8Array(asset.bytes)], {type: asset.mime}) : null); + if (disposed || !data) return; + objectUrl = URL.createObjectURL(data); setUrl(objectUrl); + }; + void load().catch(() => {}); + return () => { disposed = true; if (objectUrl) URL.revokeObjectURL(objectUrl); }; + }, [audio.assetId, blob]); + return url ? <> +