",
+ "css": ".ds-badges{display:flex;gap:6px;} .ds-badge{font-family:Manrope,ui-sans-serif,system-ui,sans-serif;display:inline-flex;align-items:center;height:20px;padding:2px 8px;border-radius:9999px;border:1px solid color-mix(in srgb,var(--primary,#3d5a86) 20%,transparent);background:color-mix(in srgb,var(--primary,#3d5a86) 10%,transparent);color:var(--primary,#3d5a86);font-size:12px;font-weight:500;} .ds-badge.ok{border-color:color-mix(in srgb,var(--good,#3f7a4f) 30%,transparent);background:color-mix(in srgb,var(--good,#3f7a4f) 10%,transparent);color:var(--good,#3f7a4f);} .ds-badge.bad{border-color:color-mix(in srgb,var(--destructive,#9e3b28) 25%,transparent);background:color-mix(in srgb,var(--destructive,#9e3b28) 10%,transparent);color:var(--destructive,#9e3b28);}"
+ }
+ ],
+ "narrative": {
+ "northStar": "The Lab Notebook on Warm Paper",
+ "overview": "openfinetuner wears the opencontroller console's system, Paper & Ink, softened, on purpose: the studio embeds inside opencontroller and has to read as the same instrument. Near-black ink sits on a white page; every neutral is warmed toward paper (no pure black, no cool greys); panels are white blocks set apart by a warm hairline rather than by depth. Corners are near-square and soften only as surfaces grow.\n\nThe density is a working console's: small type (14px body, 12px labels, 11px meta), tight but regular padding, and every number in tabular figures. Nothing is filled solid. The single interactive colour, a muted slate blue, marks what can be acted on and where you are, always as a soft tint with deeper text. Verdicts speak in earthy tints the same way: moss green for good, ochre for not-there-yet, brick for failure. Data, ids, model names and code are set in JetBrains Mono, so what is machine-truth reads differently from what is prose.\n\nThe cockpit added the system's first signature components: a Ctrl agent thread whose action cards carry one tinted primary and a quiet Edit, a four-station loop rail drawn as one continuous line, a station inspector, and a scorecard. Light and dark are equal citizens: dark is warm graphite with the same roles, and the embed host's theme wins when framed.",
+ "keyCharacteristics": [
+ "White canvas, warm hairlines, warm-brown faint shadows; depth by edge, not elevation.",
+ "One slate-blue accent, used only as a tint (10% fill, 20-30% edge, full-strength text).",
+ "Earthy verdict tints (good / warning / destructive) in the same tint grammar.",
+ "Manrope for every word, JetBrains Mono for every datum; tabular numbers throughout.",
+ "Near-square corners from one radius (8px); pills only for pills.",
+ "Motion is short and functional (150-250ms ease-out) and disappears under reduced motion."
+ ],
+ "rules": [
+ {
+ "name": "The Tint, Never Fill Rule",
+ "body": "No component fills solid with an accent or verdict colour. Primary, good, warning and destructive appear as a 5-15% background with a 20-40% edge and full-strength text. A solid slate button is off-system.",
+ "section": "colors"
+ },
+ {
+ "name": "The Warm Neutral Rule",
+ "body": "Every grey leans toward paper. No pure black text or surface, no cool grey, and in dark mode no neutral greys: graphite carries a hint of the same warmth.",
+ "section": "colors"
+ },
+ {
+ "name": "The Verdict Earns Its Colour Rule",
+ "body": "Good, warning and destructive appear only where a real state or score backs them; a decorative green or ochre is a lie on a surface that must not invent metrics.",
+ "section": "colors"
+ },
+ {
+ "name": "The Mono Is Machine-Truth Rule",
+ "body": "Anything a person could paste into a shell or an API (ids, model names, config keys, commands, logs) is JetBrains Mono; prose never is.",
+ "section": "typography"
+ },
+ {
+ "name": "The No Capitals Rule",
+ "body": "Labels are sentence case at 11-12px, weight 500. Uppercase tracked labels are not part of this system.",
+ "section": "typography"
+ },
+ {
+ "name": "The Container Not Viewport Rule",
+ "body": "Components respond to their container's width (container queries, ResizeObserver), because the studio is embedded at sizes the viewport cannot predict.",
+ "section": "layout"
+ },
+ {
+ "name": "The One Scroller Rule",
+ "body": "On a phone the page is the only scroller; nested scroll panes exist only when panes sit side by side.",
+ "section": "layout"
+ },
+ {
+ "name": "The Hairline First Rule",
+ "body": "Separate with a 1px warm hairline before reaching for a shadow; a shadow marks a floating surface or the one live card, never a resting panel.",
+ "section": "elevation"
+ }
+ ],
+ "dos": [
+ "**Do** express every accent and verdict as a tint: 5-15% fill, 15-40% edge, full-strength text.",
+ "**Do** set ids, model names, config keys, commands and logs in JetBrains Mono, and every number in tabular figures.",
+ "**Do** separate panels with a 1px warm hairline and keep the paper shadow for floating surfaces and the live action card.",
+ "**Do** derive radii from the 8px base and keep pills for badges, chips, markers and bars.",
+ "**Do** size layouts from the container (container queries at 56rem for the cockpit split), so embedded and phone widths behave.",
+ "**Do** keep motion to 150-250ms ease-out state changes, plus the dash-flow only while work runs, all off under reduced motion.",
+ "**Do** keep light and dark equal: every colour role has its warm-graphite counterpart."
+ ],
+ "donts": [
+ "**Don't** fill a button, badge or callout solid with slate or a verdict colour.",
+ "**Don't** use pure black (#000) for text or surfaces, or cool greys, in either theme.",
+ "**Don't** colour a state good, warning or destructive without a real score or status behind it.",
+ "**Don't** add uppercase tracked labels, kickers or eyebrows above titles.",
+ "**Don't** introduce a second display face; Manrope carries headings.",
+ "**Don't** add heavy or black shadows in light mode, or lift resting panels with elevation.",
+ "**Don't** give the side panel page width on a phone, or nest scroll panes when the cockpit is stacked."
+ ]
+ }
+}
\ No newline at end of file
diff --git a/.impeccable/surfaces/frontend-src-features-cockpit-cockpit-tsx.md b/.impeccable/surfaces/frontend-src-features-cockpit-cockpit-tsx.md
new file mode 100644
index 0000000..53592ec
--- /dev/null
+++ b/.impeccable/surfaces/frontend-src-features-cockpit-cockpit-tsx.md
@@ -0,0 +1,32 @@
+---
+version: 1
+slug: "frontend-src-features-cockpit-cockpit-tsx"
+primary_target: "frontend/src/features/cockpit/Cockpit.tsx"
+related_targets: ["frontend/src/features"]
+---
+
+# Fine-tuning cockpit
+
+Scope: the studio's surface for making and improving one fine-tuned model (a project). Visitor mode: Operate. Both Business and Research modes use it; Research gets full settings, logs and every number in the inspector.
+
+Audience and job: a business user turns their knowledge, examples or agent traffic into a proven model they can deploy; a researcher iterates versions on evidence. Success: from a sentence to an evaluated fine-tune in one sitting, then each next version justified by the scorecard.
+
+Constraints: the opencontroller console visual system stays (tokens, shadcn primitives, light and dark); embeddable in opencontroller; every action Ctrl agent takes is visible, reversible where possible, and has an SDK/CLI equivalent; no invented metrics.
+
+## Direction contract
+
+THESIS: Ctrl agent and the loop are one instrument: a conversation proposes each next step as an approvable action, and a live map of the whole loop (data, fine-tune, evaluate, deploy) shows the consequence the moment you approve. It refuses the category default of a settings form followed by a separate runs page.
+
+OWN-WORLD: The console's Paper & Ink: white canvas, warm hairlines, slate-blue tint for what can be acted on and what is live, earthy good/warning/destructive tints for verdicts. Action cards are bordered panels with one tinted primary and one quiet edit. Loop nodes are four hairline-framed stations on a single rail, the active one tinted, the rail drawn as one continuous line whose segment flows while work runs.
+
+STORY: The visitor says what the model should do; Ctrl agent answers with a plan they approve; the map lights station by station; the scorecard lands in the Evaluate station and Ctrl agent proposes the next version or the deploy. They leave knowing whether their model is ready and why.
+
+FIRST VIEWPORT: Left 7/12: the loop rail across the top (Data → Fine-tune → Evaluate → Deploy, each station with its status line and version chips), the selected station's inspector filling the rest (examples, live loss curve, scorecard, endpoint). Right 5/12: Ctrl agent's conversation, header with the project's name and stage, thread of Ctrl agent's messages and action cards, composer pinned at the bottom with suggested next actions as chips (stacked on a phone, Ctrl agent comes first). The primary action is always the newest action card's tinted button in the conversation.
+
+FORM: Copilot cockpit (the agent named Ctrl agent, placed on the right by the user after the build), position 1 of 7 in the ranked list (dealt second); seed key f7febb6b. Signature interaction: approving an action card lights its station and runs the rail segment into it; hovering a message highlights its station; clicking a station scrolls the thread to its latest message. Motion grammar: 150–250 ms ease-out state changes, a dash-flow on the active rail segment while a job runs, all of it off under reduced motion.
+
+FINISH: unreviewed and undocumented is unfinished; this build ends with the finish review, the verdict, DESIGN.md, and every shipping raster carrying its provenance
+
+## Unresolved
+
+- Free-text understanding is rule-based (intents and goal classification) in this build; planning with the user's frontier model is a follow-up.
diff --git a/CLAUDE.md b/CLAUDE.md
index 4339fd1..1ce4e1e 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -5,7 +5,7 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co
## What this is
ShadowLM Trainer is a fine-tuning SDK: load any open model, train it with any of
-13 methods, on any hardware, then own the weights. The headline use case is
+15 methods, on any hardware, then own the weights. The headline use case is
"shadowing" — moving one task off a rented frontier model onto a small model you
own, by capturing real agent traffic (`slm.capture()`), judging episodes, and
training on them — without modifying the agent (the model API is the only
@@ -87,14 +87,18 @@ touches one file and no others.
### The shadowing / agent-tuning loop
+There are four ways data gets in, and they all end at a `Trajectory`:
+`capture.py` (live), `traces.py` (already ran), `synth/` (doesn't exist yet), and
+`Dataset.from_*` (you have a file).
+
- `capture.py` — `slm.capture(model)` is a drop-in OpenAI-compatible proxy that
records an unmodified agent's traffic, reconstructing message-level
trajectories (calls that extend a prior call's message prefix merge into one
episode; use an `x-session-id` header to disambiguate interleaved conversations).
- `traces.py` — the offline sibling of `capture.py`: ingests OpenTelemetry GenAI
- spans (OTLP JSON, event/spec/OpenInference/raw-wire shapes), groups them into
- conversations, and `to_dataset()`s them. For agents already instrumented —
- no proxy in the path.
+ spans, groups them into conversations, and `to_dataset()`s them. Four dialects
+ are read (OTel spec `{role,parts}`, OpenInference, indexed OpenLLMetry,
+ OpenAI-wire blobs). For agents already instrumented — no proxy in the path.
- `rl.py` — `Trajectory` / `TrajectoryGroup` / `judge_group` (LLM-judge scoring),
fed into `method="grpo"`.
- `apo.py` — `optimize_prompt()`: optimize the prompt instead of weights, same
@@ -102,6 +106,38 @@ touches one file and no others.
- `eval.py` — `slm.evaluate()` / `shadowlm eval`: score a model on a held-out set
(exact / contains / numeric / JSON / LLM-judge scorers).
+### The synthesizer (`synth/`)
+
+The fourth inlet — for traffic that doesn't exist yet (cold start, amplifying a
+handful of episodes, covering cases production never hit). Two orthogonal axes
+again, mirroring backends × methods:
+
+- **seeds** (`seeds.py`) — where scenarios come from: a plain-English `task`, a
+ `document` (chunked, facts extracted, answers judged against the passage), or
+ real `episodes` to vary. Any seed composes with any mode.
+- **modes** (`generate.py`) — what gets written per scenario: a `conversation`,
+ a `preference` pair, or `paraphrases`. Generation is always **taxonomy first,
+ instances second** — that structure, not prompt wording, is what stops mode
+ collapse.
+
+Everything converges on `Trajectory` (the same type capture and traces produce),
+then `emit.py` renders it into the shape the consumer takes. **Output shape is
+chosen from the method's spec, never its name** (`resolve_output`). `to_otlp` is
+the exact inverse of `traces._spec_message` — change one and you must change the
+other; `tests/test_synth_otlp_roundtrip.py` is what holds them together.
+
+`quality.py` validates (the "must end on an assistant turn" rule is load-bearing
+— see `torch.py:_train_dataset`), deduplicates, and gates on a judge score.
+Nothing is dropped silently: `SynthReport` reconciles exactly, and
+`report.balanced` asserts it.
+
+A run costs money, so it is meterable and stoppable. Tokens come from the
+provider's own `usage` block — never estimated, and there is deliberately no
+price table to go stale. `token_budget=` and `should_stop=` end a run early
+while **keeping** what it produced; both gate *generation* only, because
+leaving already-generated rows unscored fails them at the gate and wastes the
+whole spend. Studio runs persist under `work_root/synth/` and are cancellable.
+
### Signature methods (MoRE)
`more.py` / `more_plus.py` implement "mixture of retrieval experts" — facts fused
@@ -128,13 +164,40 @@ routing; its run progress is one step per unit (see `resolve_total_steps`).
Same protocol backs `backend="remote"` and ShadowLM Studio. `SHADOWLM_API_URL`
may list several servers; `pick()` binds to the least-busy reachable one for
the session (client-side routing, deliberately not a scheduler).
-- `frontend/` — React 19 + Vite + Tailwind v4 studio. `npm run build` outputs to
+- `frontend/` — React 19 + Vite + Tailwind v4 studio. Layout: `src/app/` (shell,
+ `router.ts` — one typed route table over the URL hash, the format the embed
+ bridge speaks), `src/features//` (one folder per surface: cockpit,
+ projects, datasets, models, train, runs, evaluate, deployments, playground,
+ machines, overview), `src/components/` (ui primitives + shared pieces),
+ `src/lib/` (`queries.ts` is the react-query data layer: one hook per
+ resource, polling only while something runs). The **cockpit**
+ (`features/cockpit/`) is where a model is made: the Ctrl agent conversation
+ on the right (`copilot.ts` narrates state and proposes the next action; rule-based, no
+ model call) drives a live map of the loop on the left (`model.ts` folds every resource
+ into Data → Fine-tune → Evaluate → Deploy stations). Its direction contract
+ lives in `.impeccable/surfaces/`. `npm run build` outputs to
`../shadowlm/_static` (the wheel ships the compiled UI; end users never need
node). `frontend/src/api.ts` is the typed mirror of the remote protocol. The
pages (Dashboard · Datasets → Models → Train → Runs → Playground · Machines)
are the capture→train→own loop as a UI. Auth has three modes — `password`,
`apikey`, or `none` (`GET /v1/auth` reports which) — plus long-lived, hashed,
individually-revocable **machine tokens** that workers authenticate with.
+ The studio has two modes over the same objects (`frontend/src/lib/mode.ts`,
+ asked once on first visit, switched in the side panel): **Business** walks a
+ *project* (`/v1/projects`: one fine-tune for a job, goal knowledge / task /
+ takeover) through Data → Fine-tune → Evaluate → Deploy, with base model and
+ method picked by `lib/recipe.ts` and shown; **Research** keeps the object
+ pages (datasets, models, runs) plus **Evaluate** (`/v1/evals`: several
+ targets scored on the same questions, run as a background job and persisted
+ under `/evals/`). Chat runs as a background task too
+ (`/v1/tasks/chat`, polled by `chatAsync`), because a held request dies at the
+ proxy's 100 s. The user's **frontier model** (settings: OpenAI-compatible
+ base URL, key, model; `shadowlm/frontier.py`) is an eval baseline, the
+ `judge` metric, and the upstream for **agent capture** (`/v1/capture/`
+ passes calls through and records them; `capture.reconstruct` turns them into
+ episodes → a dataset). **Deployments** serve a fine-tune at `/openai/v1`
+ (OpenAI-compatible, a hashed per-deployment key, no studio login). Product
+ context and principles live in `PRODUCT.md`.
The UI uses the opencontroller console's design system (shadcn primitives in
`frontend/src/components/ui/`, tokens in `index.css`, light + dark) and can
run **embedded** in a host console over the `oc-embed/1` postMessage bridge
diff --git a/DESIGN.md b/DESIGN.md
new file mode 100644
index 0000000..1954c88
--- /dev/null
+++ b/DESIGN.md
@@ -0,0 +1,345 @@
+---
+name: openfinetuner
+description: The fine-tuning studio (formerly ShadowLM), in the opencontroller console's Paper & Ink, softened.
+colors:
+ canvas: "#ffffff"
+ ink-foreground: "#141311"
+ surface: "#ede9e1"
+ surface-2: "#e8e4dd"
+ subtle: "#f6f3ed"
+ side-panel: "#f8f7f4"
+ muted-foreground: "#635d51"
+ hairline: "#e3ded4"
+ hairline-strong: "#d2cdc0"
+ input-edge: "#e8e4dd"
+ slate-primary: "#3d5a86"
+ primary-foreground: "#fdfcf9"
+ good: "#3f7a4f"
+ warning: "#a5671c"
+ destructive: "#9e3b28"
+ console-ink: "#141311"
+ console-bone: "#ebe8e3"
+ dark-canvas: "#121110"
+ dark-background: "#1a1917"
+ dark-popover: "#1e1d1b"
+ dark-surface: "#201f1c"
+ dark-surface-2: "#262522"
+ dark-foreground: "#ebe8e3"
+ dark-muted-foreground: "#a39e95"
+ dark-hairline: "#2a2926"
+ dark-hairline-strong: "#38362f"
+ dark-slate-primary: "#9db2d4"
+ dark-good: "#6faa83"
+ dark-warning: "#cfa25a"
+ dark-destructive: "#d4685a"
+typography:
+ headline:
+ fontFamily: "Manrope, ui-sans-serif, system-ui, sans-serif"
+ fontSize: "1.25rem"
+ fontWeight: 600
+ lineHeight: 1.25
+ letterSpacing: "-0.015em"
+ title:
+ fontFamily: "Manrope, ui-sans-serif, system-ui, sans-serif"
+ fontSize: "1rem"
+ fontWeight: 600
+ lineHeight: 1.25
+ letterSpacing: "-0.01em"
+ figure:
+ fontFamily: "Manrope, ui-sans-serif, system-ui, sans-serif"
+ fontSize: "1.5rem"
+ fontWeight: 600
+ lineHeight: 1.333
+ letterSpacing: "-0.01em"
+ fontFeature: "\"tnum\""
+ body:
+ fontFamily: "Manrope, ui-sans-serif, system-ui, sans-serif"
+ fontSize: "0.875rem"
+ fontWeight: 400
+ lineHeight: 1.625
+ fontFeature: "\"cv11\", \"ss01\""
+ nav:
+ fontFamily: "Manrope, ui-sans-serif, system-ui, sans-serif"
+ fontSize: "13px"
+ fontWeight: 500
+ label:
+ fontFamily: "Manrope, ui-sans-serif, system-ui, sans-serif"
+ fontSize: "0.75rem"
+ fontWeight: 500
+ meta:
+ fontFamily: "Manrope, ui-sans-serif, system-ui, sans-serif"
+ fontSize: "11px"
+ fontWeight: 500
+ mono:
+ fontFamily: "JetBrains Mono, ui-monospace, SF Mono, monospace"
+ fontSize: "0.75rem"
+ fontWeight: 400
+ lineHeight: 1.625
+rounded:
+ sm: "4px"
+ md: "6px"
+ lg: "8px"
+ xl: "12px"
+ pill: "9999px"
+spacing:
+ hair: "2px"
+ xs: "4px"
+ sm: "8px"
+ md: "12px"
+ cell-y: "14px"
+ lg: "16px"
+ panel-x: "20px"
+ page: "24px"
+components:
+ button-primary:
+ backgroundColor: "color-mix(in srgb, {colors.slate-primary} 10%, transparent)"
+ textColor: "{colors.slate-primary}"
+ rounded: "{rounded.lg}"
+ padding: "0 10px"
+ height: "32px"
+ button-primary-hover:
+ backgroundColor: "color-mix(in srgb, {colors.slate-primary} 15%, transparent)"
+ button-outline:
+ backgroundColor: "{colors.canvas}"
+ textColor: "{colors.ink-foreground}"
+ rounded: "{rounded.lg}"
+ padding: "0 10px"
+ height: "32px"
+ button-outline-hover:
+ backgroundColor: "{colors.surface-2}"
+ button-ghost:
+ backgroundColor: "transparent"
+ textColor: "{colors.muted-foreground}"
+ rounded: "{rounded.lg}"
+ padding: "0 10px"
+ height: "32px"
+ button-destructive:
+ backgroundColor: "color-mix(in srgb, {colors.destructive} 10%, transparent)"
+ textColor: "{colors.destructive}"
+ rounded: "{rounded.lg}"
+ height: "32px"
+ input:
+ backgroundColor: "transparent"
+ textColor: "{colors.ink-foreground}"
+ rounded: "{rounded.lg}"
+ padding: "4px 10px"
+ height: "32px"
+ chip-suggestion:
+ backgroundColor: "{colors.canvas}"
+ textColor: "{colors.ink-foreground}"
+ typography: "{typography.label}"
+ rounded: "{rounded.md}"
+ padding: "0 8px"
+ height: "24px"
+ badge:
+ backgroundColor: "color-mix(in srgb, {colors.slate-primary} 10%, transparent)"
+ textColor: "{colors.slate-primary}"
+ typography: "{typography.label}"
+ rounded: "{rounded.pill}"
+ padding: "2px 8px"
+ height: "20px"
+ version-chip:
+ backgroundColor: "color-mix(in srgb, {colors.slate-primary} 10%, transparent)"
+ textColor: "{colors.slate-primary}"
+ rounded: "{rounded.pill}"
+ padding: "0 6px"
+ height: "20px"
+ panel:
+ backgroundColor: "{colors.canvas}"
+ rounded: "{rounded.lg}"
+ padding: "14px 16px"
+ nav-item:
+ backgroundColor: "transparent"
+ textColor: "{colors.muted-foreground}"
+ typography: "{typography.nav}"
+ rounded: "{rounded.md}"
+ padding: "0 6px"
+ height: "32px"
+ nav-item-active:
+ backgroundColor: "color-mix(in srgb, {colors.slate-primary} 10%, transparent)"
+ textColor: "{colors.slate-primary}"
+ station-marker:
+ backgroundColor: "{colors.canvas}"
+ rounded: "{rounded.pill}"
+ size: "24px"
+ log-console:
+ backgroundColor: "{colors.console-ink}"
+ textColor: "{colors.console-bone}"
+ typography: "{typography.mono}"
+ padding: "16px"
+---
+
+# Design System: openfinetuner
+
+## Overview
+
+**Creative North Star: "The Lab Notebook on Warm Paper"**
+
+openfinetuner wears the opencontroller console's system, Paper & Ink, softened, on purpose: the studio embeds inside opencontroller and has to read as the same instrument. Near-black ink sits on a white page; every neutral is warmed toward paper (no pure black, no cool greys); panels are white blocks set apart by a warm hairline rather than by depth. Corners are near-square and soften only as surfaces grow.
+
+The density is a working console's: small type (14px body, 12px labels, 11px meta), tight but regular padding, and every number in tabular figures. Nothing is filled solid. The single interactive colour, a muted slate blue, marks what can be acted on and where you are, always as a soft tint with deeper text. Verdicts speak in earthy tints the same way: moss green for good, ochre for not-there-yet, brick for failure. Data, ids, model names and code are set in JetBrains Mono, so what is machine-truth reads differently from what is prose.
+
+The cockpit added the system's first signature components: a Ctrl agent thread whose action cards carry one tinted primary and a quiet Edit, a four-station loop rail drawn as one continuous line, a station inspector, and a scorecard. Light and dark are equal citizens: dark is warm graphite with the same roles, and the embed host's theme wins when framed.
+
+**Key Characteristics:**
+- White canvas, warm hairlines, warm-brown faint shadows; depth by edge, not elevation.
+- One slate-blue accent, used only as a tint (10% fill, 20-30% edge, full-strength text).
+- Earthy verdict tints (good / warning / destructive) in the same tint grammar.
+- Manrope for every word, JetBrains Mono for every datum; tabular numbers throughout.
+- Near-square corners from one radius (8px); pills only for pills.
+- Motion is short and functional (150-250ms ease-out) and disappears under reduced motion.
+
+## Colors
+
+A warm-neutral paper palette with one slate-blue voice and three earthy verdict tints, mirrored into warm graphite for dark.
+
+### Primary
+- **Slate Primary** (`slate-primary`; dark `dark-slate-primary`): the one interactive colour. Primary buttons, the active nav row, the selected loop station, ready/running station words, version chips, the "Open in Research" link, focus rings, checkbox accent, text selection (24% mix) and the caret. Always a tint: background at 5-15%, edge at 15-40%, the token itself only for text, icons, the loss curve and the running dot.
+
+### Secondary
+- **Moss Good** (`good`): passed verdicts, done stations, correct-answer checks, the eval point on the loss curve, the "live" dot. Aliased as `success` in code.
+- **Ochre Warning** (`warning`): "not there yet" verdicts and the Evaluate station when the fine-tune beats its base but misses the bar.
+- **Brick Destructive** (`destructive`): failures, wrong-answer crosses, destructive actions, validation errors.
+
+### Neutral
+- **Canvas** (`canvas`): the page, cards and panels in light mode; content reads crisp on white.
+- **Ink Foreground** (`ink-foreground`): all body text and headings; warm near-black, never #000.
+- **Muted Foreground** (`muted-foreground`): secondary lines, labels, meta, idle stations.
+- **Surface / Surface 2** (`surface`, `surface-2`): raised fills: table heads, code backgrounds (often at 40%), segmented-control track, outline-button hover.
+- **Subtle** (`subtle`): a whisper for hover rows, zebra rows and the person's own messages in the thread.
+- **Side Panel** (`side-panel`): the floating nav island; the warmth lives here, in the fills and in the rules.
+- **Hairline / Hairline Strong** (`hairline`, `hairline-strong`): dividers and panel edges; the stronger edge for markers in a muted state and thin scrollbars.
+- **Console Ink / Bone** (`console-ink`, `console-bone`): the training-log console, dark in either theme; also the phone overlay scrim (ink at 30%).
+
+### Named Rules
+**The Tint, Never Fill Rule.** No component fills solid with an accent or verdict colour. Primary, good, warning and destructive appear as a 5-15% background with a 20-40% edge and full-strength text. A solid slate button is off-system.
+
+**The Warm Neutral Rule.** Every grey leans toward paper. No pure black text or surface, no cool grey, and in dark mode no neutral greys: graphite carries a hint of the same warmth.
+
+**The Verdict Earns Its Colour Rule.** Good, warning and destructive appear only where a real state or score backs them; a decorative green or ochre is a lie on a surface that must not invent metrics.
+
+## Typography
+
+**Body Font:** Manrope (with ui-sans-serif, system-ui)
+**Heading Font:** Manrope, the same face, so the page and side panel read as one voice
+**Mono Font:** JetBrains Mono (with ui-monospace, SF Mono)
+
+**Character:** a quiet geometric sans for every word, set small and tight, with a crisp mono for everything a machine produced. Hierarchy comes from size and a step of weight (500 to 600) plus slight negative tracking, never from heavy weights or capitals.
+
+### Hierarchy
+- **Headline** (600, 1.25rem, -0.015em, balanced wrap): page titles in the page header and the mode chooser.
+- **Title** (600, 1rem, -0.01em): panel and station headers, the cockpit's project name, action-card titles (at 0.875rem).
+- **Figure** (600, 1.5rem, tabular): scorecard scores and stat-strip values.
+- **Body** (400, 0.875rem, relaxed leading; descriptions capped at 64-70ch): Ctrl agent messages, table cells, descriptions (page descriptions at 13px).
+- **Nav** (500, 13px): side-panel rows.
+- **Label** (500, 0.75rem): buttons at xs, field labels, chips, station state lines, captions.
+- **Meta** (500, 11px): message author line, station status word, side-panel section titles, "formerly ShadowLM", version strings.
+- **Mono** (400, 0.75rem; 11px in consoles and dense metadata; 10px in the monogram and version chips): run ids, model ids, config keys, CLI blocks, logs, chart axes. Ligatures off in typed mono fields.
+
+### Named Rules
+**The Mono Is Machine-Truth Rule.** Anything a person could paste into a shell or an API (ids, model names, config keys, commands, logs) is JetBrains Mono; prose never is.
+
+**The No Capitals Rule.** Labels are sentence case at 11-12px, weight 500. Uppercase tracked labels are not part of this system.
+
+## Layout
+
+The studio shell is a floating side-panel island (240px open, 48px folded to an icon rail; 12px inset from the viewport, 24px gap to the page). Pages pad 24px on the sides and top, 32px at the bottom, and use container queries rather than viewport breakpoints, so the same page works framed inside opencontroller.
+
+Panels set their own internal rhythm: headers at 20px horizontal by 14px vertical with a hairline beneath, cells and callouts at 16px by 14px, sibling panels 16px apart, page header 28px above content. Grids of figures (stat strips, scorecard targets) are joined by a 1px hairline gap on a hairline-coloured ground, not by separate boxes.
+
+The cockpit is a 12-column split at 56rem of its own width: conversation 5/12 on the left, loop rail and inspector 7/12 on the right, each pane scrolling inside the viewport. Below that width the panes stack, flow at natural height and the page becomes the only scroller; the composer goes sticky at the bottom and earlier messages fold behind "Show N earlier". The loop rail runs four columns wide and two columns (breaking after Fine-tune) when narrow. Tables of answers become stacked blocks below the xl container width.
+
+Below 768px the side panel never takes page width: it folds to the 48px rail and opens as an overlay above an ink scrim at 30% with a 1px blur.
+
+### Named Rules
+**The Container Not Viewport Rule.** Components respond to their container's width (container queries, ResizeObserver), because the studio is embedded at sizes the viewport cannot predict.
+
+**The One Scroller Rule.** On a phone the page is the only scroller; nested scroll panes exist only when panes sit side by side.
+
+## Elevation & Depth
+
+The system is flat with a hairline edge. Panels are separated from the canvas by a 1px warm hairline; a faint warm-brown shadow is reserved for the few things that float or that the eye should land on first: the side-panel island, the active segment of the mode switch, and the live action card. Popovers, selects, tooltips and sheets take the raised shadow. In dark mode both shadows switch to black at higher opacity, and depth comes mostly from canvas (darkest) to card to hover (lighter).
+
+### Shadow Vocabulary
+- **Paper** (`box-shadow: 0 1px 0 rgb(28 24 18 / 0.03), 0 1px 2px rgb(28 24 18 / 0.05)`): the side-panel island, the selected mode segment, the live action card.
+- **Raised** (`box-shadow: 0 1px 0 rgb(28 24 18 / 0.04), 0 2px 4px rgb(28 24 18 / 0.06), 0 8px 24px -12px rgb(28 24 18 / 0.1)`): popovers, menus, select content, tooltips, sheets.
+
+### Named Rules
+**The Hairline First Rule.** Separate with a 1px warm hairline before reaching for a shadow; a shadow marks a floating surface or the one live card, never a resting panel.
+
+## Shapes
+
+Every radius derives from one 8px base: 4px for tiny inline targets, 6px for compact controls, chips, nav rows and loop stations, 8px for buttons, inputs and any box drawn with a full border (applied globally), 12px for the floating side panel. Pills are reserved for true pills: badges, version chips, the station markers, progress bars and status dots. Borders are always 1px; dashed hairlines mark empty or spent states (a spent proposal, an empty chart grid). Tables framed by a border clip their rows to its corners.
+
+## Components
+
+### Buttons
+Tinted and quiet; a button announces it can be acted on without shouting.
+- **Shape:** gently squared (8px); 32px tall by default, 24px (xs) and 28px (sm) compact sizes at 6px.
+- **Primary:** slate tint: 10% slate background, 20% slate edge, slate text, 14px weight 500; hover deepens to 15%.
+- **Outline:** card background, hairline edge, ink text; hover fills with surface-2. Used for suggestion chips and secondary actions.
+- **Ghost:** no edge or fill, muted text; hover fills muted. The quiet Edit and Copy buttons.
+- **Destructive:** brick tint at 10%, hover 20%.
+- **Link:** slate text, underline on hover (the "Open in Research" pattern).
+- **Hover / Focus:** colour transitions; focus shows the ring edge plus a 3px slate ring at 50%; press nudges down 1px.
+
+### Chips and Badges
+- **Suggestion chips:** outline xs buttons (24px, weight 400) above the composer; wrap on wide, one horizontally scrolling row when stacked.
+- **Badges:** 20px pills in the tint grammar (primary, secondary surface, destructive, outline). Run status badges use good/warning/destructive tints at 10% with a 30% edge.
+- **Version chips:** 20px mono pills on the Fine-tune station; the current version tinted slate, others hairline and muted.
+
+### Cards / Containers
+- **Corner Style:** 8px for bordered panels.
+- **Background:** canvas (card), with surface at 40% for neutral callouts and code bodies.
+- **Shadow Strategy:** none at rest (see Elevation & Depth).
+- **Border:** 1px hairline.
+- **Internal Padding:** 16px by 14px for callouts and cells; 20px for station and thread content.
+
+### Inputs / Fields
+- **Style:** 1px input-edge stroke, transparent fill, 8px radius, 32px tall (composer textarea grows from 36px to 128px). Bare native controls get the same treatment globally.
+- **Focus:** edge turns slate, 3px slate ring at 50%.
+- **Error / Disabled:** brick edge with a 20% brick ring; disabled at 50% opacity. Field labels sit above at 12px muted.
+
+### Navigation
+- **Side-panel island:** floating, side-panel fill, 12px radius, paper shadow. Brand row is the "of" mono monogram (28px, slate tint, 6px radius) with "openfinetuner" over "formerly ShadowLM" at 11px.
+- **Rows:** 32px, 13px weight 500, muted with a 70%-opacity 16px line icon; hover fills surface-2 at 70%; active is the slate tint with full-opacity icon.
+- **Section titles:** 11px muted labels; folded, they become a centred hairline.
+- **Mode switch:** a two-segment radio (Business / Research) on a surface-2 track; the selected segment is card-white with the paper shadow. Folded, it becomes a single icon button.
+- **Phone:** the island folds to the 48px rail and opens as an overlay over the ink scrim.
+
+### Ctrl Agent Thread (signature)
+- **Messages:** an 11px meta author line ("Ctrl agent" with a 12px line icon) and, at its right, the station it speaks about as a small link; text at 14px relaxed. The person's own messages are right-aligned blocks on subtle with a hairline, max 85% wide.
+- **Linking:** hovering a message tints its station on the rail; new messages rise in (5px, 180ms).
+- **Action card:** the one live proposal is a bordered panel with the paper shadow: a 14px semibold title, a two-column definition list (muted terms), "why" lines with a middle-dot, one tinted primary action and a ghost Edit, and a "Same run from your shell" disclosure revealing the CLI in mono on surface. Spent proposals collapse to a dashed-hairline line.
+- **Live line:** while work runs, a strip at 5% slate above the composer with a flowing dash.
+
+### Loop Rail (signature)
+- Four stations (Data, Fine-tune, Evaluate, Deploy) on one continuous 2px line. Each marker (24px circle, card fill) sits alone on the line; the station's name (14px semibold), status word (11px, toned) and line (12px muted) hang below.
+- **Segment states:** hairline when ahead, slate at 40% after a done station, moss at 60% between done stations, a slate dash flowing (600ms linear loop) into a running station.
+- **Station states:** idle dashed-circle muted, ready slate dot, running pinging slate dot, done moss check, Evaluate verdicts ochre alert or muted minus, failed brick cross.
+- **Selection:** selected station takes a 5% slate fill with a 30% edge; the hovered-from-thread station a 3% fill with a 15% edge. Arrow keys move between stations.
+
+### Station Inspector and Scorecard (signature)
+- **Station header:** 16px semibold title, 14px muted state line, an inline 12px "Open in Research" slate text link beneath, actions at the right; hairline below.
+- **Scorecard:** a verdict callout in its tint (5% fill, 30% edge, toned semibold title, muted detail); per-target figures joined by hairline gaps (24px tabular score, "of N correct", a 6px pill progress bar in slate for the fine-tune and muted for comparisons, the model id in 11px mono); answers as a fixed table when wide, stacked blocks when narrow, each answer prefixed by a moss check or brick cross.
+- **Log console:** ink background, bone text at 90%, 11px mono, max 288px tall with a thin scrollbar.
+
+## Do's and Don'ts
+
+### Do:
+- **Do** express every accent and verdict as a tint: 5-15% fill, 15-40% edge, full-strength text.
+- **Do** set ids, model names, config keys, commands and logs in JetBrains Mono, and every number in tabular figures.
+- **Do** separate panels with a 1px warm hairline and keep the paper shadow for floating surfaces and the live action card.
+- **Do** derive radii from the 8px base and keep pills for badges, chips, markers and bars.
+- **Do** size layouts from the container (container queries at 56rem for the cockpit split), so embedded and phone widths behave.
+- **Do** keep motion to 150-250ms ease-out state changes, plus the dash-flow only while work runs, all off under reduced motion.
+- **Do** keep light and dark equal: every colour role has its warm-graphite counterpart.
+
+### Don't:
+- **Don't** fill a button, badge or callout solid with slate or a verdict colour.
+- **Don't** use pure black (#000) for text or surfaces, or cool greys, in either theme.
+- **Don't** colour a state good, warning or destructive without a real score or status behind it.
+- **Don't** add uppercase tracked labels, kickers or eyebrows above titles.
+- **Don't** introduce a second display face; Manrope carries headings.
+- **Don't** add heavy or black shadows in light mode, or lift resting panels with elevation.
+- **Don't** give the side panel page width on a phone, or nest scroll panes when the cockpit is stacked.
diff --git a/Makefile b/Makefile
index 60b0a01..a32762f 100644
--- a/Makefile
+++ b/Makefile
@@ -30,6 +30,18 @@ help: ## list the available targets
$(PY):
python3 -m venv $(VENV)
+# Targets that run the CLI need the package *installed*, not just a venv. Guard
+# them so a fresh clone gets told what to do instead of the bare
+# "make: .venv/bin/shadowlm: No such file or directory".
+$(SHADOWLM):
+ @echo "shadowlm isn't installed in $(VENV) yet. Install it:"
+ @echo " make install # Apple Silicon (adds the mlx backend)"
+ @echo " make install-torch # CUDA or CPU"
+ @echo ""
+ @echo "Or run it from an environment you already have:"
+ @echo " python3 -m shadowlm.serve --port $(PORT)"
+ @exit 1
+
.PHONY: install
install: $(PY) ## editable install with the CLI + a training backend (mlx)
$(PIP) install -q -e '.[mlx,cli]'
@@ -44,21 +56,21 @@ frontend: ## install + build the React studio into shadowlm/_static
# ---- run --------------------------------------------------------------------
.PHONY: serve
-serve: ## run the studio + API on one port (make serve PORT=8329)
+serve: | $(SHADOWLM) ## run the studio + API on one port (make serve PORT=8329)
$(SHADOWLM) serve --port $(PORT)
.PHONY: dev
-dev: ## serve with Vite hot-reload UI alongside the backend
+dev: | $(SHADOWLM) ## serve with Vite hot-reload UI alongside the backend
$(SHADOWLM) serve --port $(PORT) --dev
.PHONY: demo
-demo: ## end-to-end smoke: a tiny finetune through the CLI
+demo: | $(SHADOWLM) ## end-to-end smoke: a tiny finetune through the CLI
$(SHADOWLM) finetune examples/sample_dataset.jsonl \
--model mlx-community/Qwen2.5-0.5B-Instruct-4bit --method lora --max-steps 8
# ---- checks -----------------------------------------------------------------
.PHONY: check
-check: ## compile the package + typecheck the frontend
+check: | $(PY) ## compile the package + typecheck the frontend
$(PY) -m compileall -q shadowlm
cd frontend && npx tsc -b
@@ -68,7 +80,7 @@ test: $(PY) ## the CPU test suite (what CI runs; tests/gpu needs a GPU box)
$(PY) -m pytest tests/ --ignore=tests/gpu -q
.PHONY: gpu-test
-gpu-test: ## the CUDA verification suite (run on a GPU box)
+gpu-test: | $(PY) ## the CUDA verification suite (run on a GPU box)
$(PY) tests/gpu/test_cuda.py
# ---- gpu (cloud demo box) ---------------------------------------------------
diff --git a/PRODUCT.md b/PRODUCT.md
new file mode 100644
index 0000000..bf8b836
--- /dev/null
+++ b/PRODUCT.md
@@ -0,0 +1,64 @@
+# Product
+
+
+
+## Platform
+
+web
+
+## Users
+
+Two audiences, served by two explicit modes of the same studio (confirmed):
+
+- **Business users**: product, operations and AI leads at enterprises (Lyzr customers and teams like the ones Lyzr demos to). They want a model of their own for a job: one that knows their company's knowledge, does a specific task from their examples, or takes over a task their agent currently sends to a rented frontier model. They want it proven before they trust it, and they think in tasks, quality, cost and "can I ship it", not in methods or learning rates.
+- **ML researchers / engineers**: people who choose base models and training methods, tune hyperparameters, read loss curves and logs, compare checkpoints and runs, and need every run reproducible from the SDK or CLI.
+
+## Product Purpose
+
+openfinetuner (formerly ShadowLM) is a fine-tuner: load any open model, fine-tune it with any of 13 methods on any hardware, evaluate it, and own the weights. Success for a business user is a fine-tuned model that does their job well enough to ship, proven on their own data; success for a researcher is a fast, inspectable, reproducible fine-tuning loop.
+
+## Positioning
+
+One fine-tuner across the whole matrix: any open model × 13 training methods (from LoRA and QLoRA to DPO, GRPO and MoRE for exact fact recall) × any hardware (CUDA, Apple Silicon, or a fleet of NAT'd workers), with every choice visible and reproducible from the SDK and CLI, and the weights staying with the user.
+
+Shadowing is one concept the fine-tuner supports, not its identity: capturing an agent's real traffic through an OpenAI-compatible proxy (`slm.capture()`) or its OpenTelemetry traces, and fine-tuning on it so a task can move off a rented frontier model without modifying the agent.
+
+## Operating Context
+
+- The studio is the web UI of `shadowlm serve`: one machine trains for real (CUDA/torch in production, mlx on Apple Silicon for development), and NAT'd machines join as workers over one outbound websocket.
+- Production runs at studio.shadowlm.sh on a single L40S 48 GB GPU box; demos to enterprise prospects (e.g. a bank's fine-tuning team) happen live in it.
+- The studio can be embedded in opencontroller (Lyzr's agent control plane) over the `oc-embed/1` bridge, where opencontroller supplies menu, sign-in, theme and role (viewer / operator / admin).
+- Everything in the UI has an SDK and CLI equivalent (`slm.load → finetune → generate → save`, `shadowlm finetune`); the Train page already shows the equivalent CLI command.
+
+## Capabilities and Constraints
+
+- In the UI today: datasets (upload JSONL, Hugging Face import, preview), a model catalog with downloads, a 4-step training wizard (data → model → method → tune), runs with live metrics, logs, checkpoints and adapter download, a Playground with fine-tune-vs-base comparison and checkpoint selection, and machines/worker tokens.
+- Agent capture, trace import, frontier comparison and serving are in the studio too: an agent's traffic passes through `/v1/capture/` to the user's frontier model and is recorded into a dataset; OpenTelemetry GenAI traces import as a dataset; evaluations can include the frontier model as a baseline and use it as the judge; and a fine-tune is served at `/openai/v1` (OpenAI-compatible, its own key per deployment). Prompt optimization (`apo.py`) remains SDK-only.
+- Training methods: 13, declared as specs (LoRA, QLoRA, DoRA, full, CPT, DPO, GRPO, MoRE, MoRE+, and others); MoRE is the method for exact fact recall.
+- One local training job at a time; inference models are cached on the GPU and freed before training.
+- Requests through Cloudflare time out after 100 seconds; long operations must not depend on a single held request.
+- Terminology: the product's unit is a fine-tune (a trained adapter over a base model). "Shadow" is legacy wording from the shadowing concept; user-facing language should say fine-tune or model, and reserve "shadow" for the agent-capture path and the compare-with-base view.
+- Shipping: a fine-tuned model is served at an OpenAI-compatible endpoint (`/openai/v1/chat/completions`) under its deployment's name and key, so an agent switches by changing base URL, key and model name; adapter download stays available. Serving shares the studio's one GPU with training and the Playground, one request at a time.
+- Frontier comparison and LLM-judge scoring (decided) use a frontier-model API key the user adds, stored on the server like the Hugging Face token. Without one, evaluation uses rule-based scorers (exact / contains / numeric / JSON) against the dataset's own answers.
+
+## Brand Commitments
+
+- Name: **openfinetuner**, shown with "formerly ShadowLM". The Python package, CLI and imports remain `shadowlm`.
+- Visual system: the opencontroller console's design system (Paper & Ink tokens, light and dark, shadcn primitives in `frontend/src/components/ui/`). This redesign keeps it in place.
+- The red brain mark stays (`shadowlm/_assets/logo.png`; white `logo1.png` in dark mode).
+- Credited as "from Lyzr Research Labs".
+- Never name competitor fine-tuning products in code, docs or UI.
+
+## Evidence on Hand
+
+- Real runs on the production box (e.g. `lyzr-assistant`, Qwen3-8B MoRE; earlier `lyzr-support`, `shadow-support`).
+- Datasets: `lyzr-faq` (10 facts about Lyzr), `lyzr-support` (50 chat rows), and bundled starter datasets.
+- No customer testimonials, benchmarks, cost-savings figures or case studies exist; the UI must not invent them. Any cost or quality comparison must come from the user's own runs.
+
+## Product Principles
+
+1. **Fine-tuning is the product.** Every journey is a fine-tune: data in, a trained model out. Shadowing is one way to bring data, not the frame.
+2. **Prove before you ship.** A fine-tune isn't done at "training complete"; it's done when an evaluation on the user's own data shows it beats its base (and the frontier model, when compared).
+3. **One studio, two depths.** Business and Research modes work on the same datasets, runs and models; switching modes never hides or forks the truth.
+4. **Nothing hidden.** Any configuration the UI chooses is visible and reproducible as SDK or CLI; no silent magic.
+5. **You own the weights.** Every successful journey ends with something the user can take away or plug in.
diff --git a/README.md b/README.md
index 4dbbb51..3b5ebd5 100644
--- a/README.md
+++ b/README.md
@@ -5,7 +5,7 @@
-
+
@@ -31,8 +31,9 @@ print(model.generate("What is the capital of France?")) # inference
model.save("out/", fmt="adapter") # ship it
```
-Change `method="lora"` to `qlora`, `dora`, `full`, `dpo`, `grpo`, `more`, `bitfit`,
-`prompt`, `ptuning`, `adapter`, `cpt`, `more_plus` — and nothing else changes. That's the idea.
+Change `method="lora"` to `qlora`, `dora`, `full`, `dpo`, `grpo`, `sdft`, `sdpo`,
+`more`, `bitfit`, `prompt`, `ptuning`, `adapter`, `cpt`, `more_plus` — and nothing else
+changes. That's the idea.
## What ShadowLM is for
@@ -65,6 +66,36 @@ run = model.finetune([group], method="grpo") # 3. train the shadowLM on them
No reward math, no rewriting the agent into an RL framework — the model API is
the one boundary every agent already has, so ShadowLM trains from it.
+## No data yet? Synthesize it
+
+Capture needs a running agent and traces need one that already ran. When you
+have neither, describe the task — a teacher model writes the training set:
+
+```python
+run = slm.synthesize( # or document="handbook.md",
+ task="Triage billing emails: classify urgency, draft a reply, " # or episodes=[...]
+ "escalate refunds over $200.",
+ teacher=slm.synth.frontier("gpt-4o"), # or any slm.load(...) model
+ n=200, method="lora") # the method picks the shape
+print(run.report.summary()) # kept 200/243 · 18 invalid · 19 dup · 6 low-scoring
+model.finetune(run.dataset, method="lora")
+```
+
+The teacher expands your task into distinct scenarios before writing anything,
+so you get coverage instead of one example rewritten 200 times. Every row is
+validated, deduplicated and judged, and **every rejection is counted** — the
+report reconciles exactly.
+
+`method=` picks the output shape from that method's spec: `dpo` gives preference
+pairs, `grpo` gives scored trajectory groups, `more_plus` gives query-diverse
+paraphrase units. `format="otlp"` emits OpenTelemetry GenAI spans that
+round-trip through `traces.from_otlp` — so the output injects into any stack
+that speaks OTel, not just this one.
+
+A run calls a paid API in a loop, so it reports the tokens the provider actually
+billed (never an estimate), and takes a `token_budget=` throttle and a
+`should_stop=` cancel. Both keep whatever the run has already produced.
+
## What you get today
The whole **capture → judge → train → own a shadowLM** loop runs on these:
@@ -73,7 +104,8 @@ The whole **capture → judge → train → own a shadowLM** loop runs on these:
|-------|--------------|-----|
| **Capture proxy** | drop-in OpenAI endpoint that records your agent's traffic into trajectories — agent unchanged | `slm.capture()` |
| **Trace ingestion** | already have OpenTelemetry GenAI spans? turn an OTLP dump into a training set, no proxy | `slm.traces.to_dataset()` |
-| **13 methods** | LoRA · QLoRA · DoRA · full · CPT · DPO · GRPO · MoRE · MoRE+ · BitFit · prompt · p-tuning · adapter | `method=` |
+| **Data synthesizer** | no traffic yet? describe the task, point at a document, or amplify a few real episodes — emitted in the shape your method takes, or as OTel spans | `slm.synthesize()` |
+| **15 methods** | LoRA · QLoRA · DoRA · full · CPT · DPO · GRPO · SDFT · SDPO · MoRE · MoRE+ · BitFit · prompt · p-tuning · adapter | `method=` |
| **Judge → train** | score episodes with an LLM judge, train with trajectory-GRPO or DPO | `judge_group` |
| **APO** | optimize the *prompt* instead of weights — same capture/judge front end, no GPU | `slm.optimize_prompt()` |
| **VERL RL** | production multi-GPU GRPO (vLLM rollouts + FSDP) for cluster-scale RL | `backend="verl"` |
@@ -102,6 +134,8 @@ spec (adapter kind, base requirements, data rendering), never the method name.
| `cpt` | continued pretraining on raw domain text | either | 5e-5 |
| `dpo` | preference optimization on `{prompt, chosen, rejected}` | either | 5e-6 |
| `grpo` | RL from reward functions or scored `TrajectoryGroup`s | either | 5e-6 |
+| `sdft` | **on-policy self-distillation** — the demo-conditioned model teaches itself; learns without forgetting | either | 1e-5 |
+| `sdpo` | **RL via self-distillation** — the feedback-conditioned self-teacher densely rescores each rollout | either | 1e-5 |
| `more` | **mixture of retrieval experts** — facts fused into attention | either | 1e-4 |
| `more_plus` | **decoupled MoE** — per-fact final-FFN LoRA experts, BM25+semantic routed, cache-safe merge | **unquantized** | 1e-4 |
| `bitfit`| train only the bias terms (~0.1% of params) | **unquantized** | 5e-4 |
@@ -178,6 +212,7 @@ Run output (mlx, a 0.5B model, ~3.5s):
## CLI & studio
```bash
+shadowlm synth --task "triage billing email" --teacher gpt-4o -n 200 -o data.jsonl
shadowlm finetune data.jsonl --model Qwen/Qwen2.5-0.5B-Instruct --method lora
shadowlm finetune --config run.yaml --dry-run # reproducible runs, preview first
shadowlm chat out/adapter/ # talk to what you trained
@@ -187,8 +222,8 @@ shadowlm serve # studio UI + API on one port
Headline hyperparameters are typed flags; every other `TrainConfig` field is
reachable via `--set field=value` or a `--config` file (flags override config
override defaults). `shadowlm serve` opens the **studio** at `http://127.0.0.1:8329`
-— Datasets (upload + HuggingFace) → Models → guided Train → live Runs (loss
-charts + training console) → Playground (compare base ↔ finetuned). It's the
+— Datasets (upload + HuggingFace + synthesize) → Models → guided Train → live
+Runs (loss charts + training console) → Playground (compare base ↔ finetuned). It's the
built React app, shipped in the wheel; the same JSON protocol powers
`backend="remote"`.
@@ -232,8 +267,8 @@ API — nothing reimplemented — to turn the blocks into a one-click migration:
```
[x] SDK — datasets → finetune → inference on mlx / torch / remote
-[x] 13 methods incl. MoRE, MoRE+ (decoupled MoE), trajectory GRPO, judge rewards
-[x] Capture proxy · OTLP trace ingestion · shadow accelerator · any-hardware
+[x] 15 methods incl. SDFT (self-distillation), SDPO (RL via self-distillation), MoRE, MoRE+ (decoupled MoE), trajectory GRPO, judge rewards
+[x] Capture proxy · OTLP trace ingestion · data synthesizer · shadow accelerator · any-hardware
[x] Remote backend + reference server + the studio dashboard + CLI
[x] Eval scorers (`slm.evaluate`, `shadowlm eval`) · worker fleet (`shadowlm worker`)
[ ] Studio orchestration — decision inbox · cost gates · shadow router · switch
diff --git a/examples/README.md b/examples/README.md
index a665ff9..58c7e03 100644
--- a/examples/README.md
+++ b/examples/README.md
@@ -29,6 +29,8 @@ python examples/remote/grpo.py
| `cpt` | ✅ | ✅ | ✅ | raw text | — |
| `dpo` | ✅ | ✅ | ✅ | preference pairs | — |
| `grpo` | ✅ | ✅ | ✅ | prompts + reward fn | — |
+| `sdft` | ✅ | ✅ | ✅ | chat | — |
+| `sdpo` | ✅ | ✅ | ✅ | prompts + reward fn | — |
| `more` | ✅ | ✅ | ✅ | facts | — |
| `more_plus` | ✅ | ✅ | ✅ | facts | unquantized |
| `bitfit` | ✅ | ✅ | ✅ | chat | unquantized + bias params |
@@ -46,18 +48,38 @@ Notes:
bitfit has nothing to train there — the examples note this and point you to a
base that has biases (e.g. `Qwen/Qwen2.5-7B-Instruct`).
+## Getting the data in
+
+The method examples above start from a file. These four start from wherever your
+data actually is — or from nothing at all:
+
+| script | starts from | ends at |
+|--------|-------------|---------|
+| `synthesize_from_task.py` | a plain-English task description | chat rows → `lora` |
+| `synthesize_from_doc.py` | a reference document | grounded paraphrase units → `more_plus` |
+| `shadow_from_traces.py` | an OTLP export of production spans | chat rows → `lora` |
+| `evaluate.py` | a trained model | a task-quality score |
+
+The two synthesis scripts call a frontier teacher, so they need
+`OPENAI_API_KEY` — or swap in `slm.synth.as_teacher(slm.load(...))` to keep the
+whole loop local.
+
## Shared data
The `data/` folder holds tiny sample datasets so the examples are self-contained:
| file | format | used by |
|------|--------|---------|
-| `data/chat.jsonl` | chat (`messages`) | lora, qlora, dora, full, bitfit, prompt, ptuning, adapter |
+| `data/chat.jsonl` | chat (`messages`) | lora, qlora, dora, full, sdft, bitfit, prompt, ptuning, adapter |
| `data/preference.jsonl` | preference (`prompt/chosen/rejected`) | dpo |
| `data/domain.jsonl` | raw text (`text`) | cpt |
| `data/facts.jsonl` | instruction (`instruction/output`) | more, more_plus |
+| `data/handbook.md` | prose | synthesize_from_doc |
+| `data/agent_traces.otlp.json` | OTel GenAI spans | shadow_from_traces |
-`grpo` defines its prompts and reward function inline in each script.
+`grpo` and `sdpo` define their prompts and reward function inline in each
+script (an `sdpo` reward fn may return `(score, feedback)` pairs — the feedback
+becomes the self-teacher's in-context signal).
There's also `shadowlm_qa.jsonl` — a chat dataset *about ShadowLM itself*, handy
for a quick end-to-end finetune that teaches a small model to answer questions
diff --git a/examples/data/handbook.md b/examples/data/handbook.md
new file mode 100644
index 0000000..1a9f4be
--- /dev/null
+++ b/examples/data/handbook.md
@@ -0,0 +1,38 @@
+# Northwind Labs — Employee Handbook (excerpt)
+
+## Expenses
+
+Expenses are submitted through the Expensify workspace within 30 days of the
+purchase. Anything at or under $75 is auto-approved. Above $75 it routes to your
+manager, and above $2,000 it also needs a director sign-off. Receipts are
+required for every line item over $25. Reimbursements land in the payroll run
+following approval, which is the 15th and the last day of each month.
+
+## Travel
+
+Book flights through the Navan portal. Economy is the default; premium economy
+is allowed on flights over six hours, and business class needs VP approval
+regardless of duration. The nightly hotel cap is $280 in New York, London and
+San Francisco, and $190 everywhere else. Rental cars are reimbursed only when
+they are cheaper than the equivalent rideshare trips.
+
+## Time off
+
+Full-time staff accrue 1.75 vacation days per month, capped at 30 accrued days.
+Unused days above the cap stop accruing rather than being forfeited. Sick leave
+is separate and untracked. Requests of five consecutive days or more should be
+filed at least three weeks ahead so the team can plan around them.
+
+## Equipment
+
+Every engineer gets a laptop refresh every three years, or sooner if the machine
+fails a diagnostic check from IT. Monitors, keyboards and chairs come out of a
+$1,200 home-office budget that resets every two years. Personal phone plans are
+not reimbursed; a company line is available on request for on-call staff.
+
+## Security
+
+Production access requires hardware two-factor authentication — TOTP apps are
+not accepted for production. Access reviews run quarterly and anything unused
+for 90 days is revoked automatically. Customer data may never be copied to a
+personal device, including for debugging.
diff --git a/examples/mlx/sdft.py b/examples/mlx/sdft.py
new file mode 100644
index 0000000..868fc38
--- /dev/null
+++ b/examples/mlx/sdft.py
@@ -0,0 +1,23 @@
+"""sdft · mlx backend
+
+SDFT — on-policy self-distillation: the model samples its own answers and is
+pulled toward itself reading the golden response in-context, so it learns the
+task with far less forgetting than SFT. Steps are slower than lora (each one
+rolls out completions). Holds a second frozen copy of the base as the teacher.
+Run from the repo root:
+ python examples/mlx/sdft.py
+"""
+import shadowlm as slm
+
+
+def main():
+ ds = slm.Dataset.from_jsonl("examples/data/chat.jsonl")
+ model = slm.load("mlx-community/Qwen2.5-0.5B-Instruct-bf16", backend="mlx")
+ run = model.finetune(ds, method="sdft", max_steps=30,
+ sdft_max_completion_length=64)
+ print("final loss:", run.loss, run.sparkline())
+ model.save("out/mlx_sdft", fmt="adapter")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/examples/mlx/sdpo.py b/examples/mlx/sdpo.py
new file mode 100644
index 0000000..f421e8f
--- /dev/null
+++ b/examples/mlx/sdpo.py
@@ -0,0 +1,36 @@
+"""sdpo · mlx backend
+
+SDPO — RL via self-distillation: the feedback-conditioned model teaches itself.
+Reward fns may return (score, feedback) pairs; the feedback (and any successful
+sibling rollout) becomes the self-teacher's in-context signal.
+Run from the repo root:
+ python examples/mlx/sdpo.py
+"""
+import shadowlm as slm
+
+
+prompts = [
+ {"prompt": "What port does the ShadowLM studio serve on? Answer with just the number.", "answer": "8329"},
+ {"prompt": "Which backend is ShadowLM's production training path? Answer with one word.", "answer": "torch"},
+ {"prompt": "What does the M in MoRE stand for? Answer with one word.", "answer": "mixture"},
+]
+
+# 1.0 on a hit; on a miss, (0.0, hint) — the hint reaches the self-teacher.
+def reward(prompts, completions, answer=None, types=None, **kwargs):
+ golds = answer if isinstance(answer, list) else [answer] * len(completions)
+ return [1.0 if (g or "").lower() in c.lower()
+ else (0.0, f"A correct answer contains {g!r}.")
+ for c, g in zip(completions, golds)]
+
+
+def main():
+ model = slm.load("mlx-community/Qwen2.5-0.5B-Instruct-bf16", backend="mlx")
+ run = model.finetune(prompts, method="sdpo", reward_fns=[reward],
+ max_steps=30, sdpo_group_size=4,
+ sdpo_max_completion_length=64)
+ print("final loss:", run.loss, run.sparkline())
+ model.save("out/mlx_sdpo", fmt="adapter")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/examples/remote/sdft.py b/examples/remote/sdft.py
new file mode 100644
index 0000000..512f55d
--- /dev/null
+++ b/examples/remote/sdft.py
@@ -0,0 +1,26 @@
+"""sdft · remote backend
+
+SDFT — on-policy self-distillation from demonstrations: learns the task from
+plain chat rows with far less forgetting than SFT.
+
+Point SHADOWLM_API_URL at your server (defaults to http://127.0.0.1:8329).
+Run from the repo root:
+ python examples/remote/sdft.py
+"""
+import os
+import shadowlm as slm
+
+os.environ.setdefault("SHADOWLM_API_URL", "http://127.0.0.1:8329")
+
+
+def main():
+ ds = slm.Dataset.from_jsonl("examples/data/chat.jsonl")
+ model = slm.load("Qwen/Qwen3-8B", backend="remote")
+ run = model.finetune(ds, method="sdft", max_steps=60,
+ sdft_max_completion_length=128)
+ print("final loss:", run.loss, run.sparkline())
+ model.save("out/remote_sdft", fmt="adapter")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/examples/remote/sdpo.py b/examples/remote/sdpo.py
new file mode 100644
index 0000000..1b69607
--- /dev/null
+++ b/examples/remote/sdpo.py
@@ -0,0 +1,39 @@
+"""sdpo · remote backend
+
+SDPO — RL via self-distillation: the feedback-conditioned model teaches itself.
+Reward fns may return (score, feedback) pairs; the feedback (and any successful
+sibling rollout) becomes the self-teacher's in-context signal.
+
+Point SHADOWLM_API_URL at your server (defaults to http://127.0.0.1:8329).
+Run from the repo root:
+ python examples/remote/sdpo.py
+"""
+import os
+import shadowlm as slm
+
+os.environ.setdefault("SHADOWLM_API_URL", "http://127.0.0.1:8329")
+
+
+prompts = [
+ {"prompt": "What port does the ShadowLM studio serve on? Answer with just the number.", "answer": "8329"},
+ {"prompt": "Which backend is ShadowLM's production training path? Answer with one word.", "answer": "torch"},
+ {"prompt": "What does the M in MoRE stand for? Answer with one word.", "answer": "mixture"},
+]
+
+# 1.0 on a hit; on a miss, (0.0, hint) — the hint reaches the self-teacher.
+def reward(prompts, completions, answer=None, types=None, **kwargs):
+ golds = answer if isinstance(answer, list) else [answer] * len(completions)
+ return [1.0 if (g or "").lower() in c.lower()
+ else (0.0, f"A correct answer contains {g!r}.")
+ for c, g in zip(completions, golds)]
+
+
+def main():
+ model = slm.load("Qwen/Qwen3-8B", backend="remote")
+ run = model.finetune(prompts, method="sdpo", reward_fns=[reward], max_steps=30)
+ print("final loss:", run.loss, run.sparkline())
+ model.save("out/remote_sdpo", fmt="adapter")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/examples/synthesize_from_doc.py b/examples/synthesize_from_doc.py
new file mode 100644
index 0000000..70564bb
--- /dev/null
+++ b/examples/synthesize_from_doc.py
@@ -0,0 +1,47 @@
+"""turn a document into a model that knows it — grounded, with MoRE+ routing
+
+Point the synthesizer at reference material and it pulls out the facts, then
+writes several differently-worded questions for each one. That phrasing variety
+is the point: MoRE+ trains one expert per fact and routes to it with BM25 over
+the *question* side, so a fact asked about only one way is an expert nobody can
+reach. Answers are judged against the source passage, so the teacher can't
+quietly invent things the document never said.
+
+Run from the repo root:
+ OPENAI_API_KEY=sk-... python examples/synthesize_from_doc.py
+"""
+from pathlib import Path
+
+import shadowlm as slm
+
+DOC = Path(__file__).resolve().parent / "data" / "handbook.md"
+PARAPHRASES_PER_FACT = 4
+
+
+def main():
+ run = slm.synthesize(
+ document=DOC,
+ teacher=slm.synth.frontier("gpt-4o"),
+ n=80,
+ method="more_plus", # → paraphrase units, one per fact
+ per_scenario=PARAPHRASES_PER_FACT,
+ )
+ print(run.report.summary()) # the note tells you the group size
+
+ # The rows come out grouped: PARAPHRASES_PER_FACT consecutive rows per fact,
+ # which is exactly what more_plus_group_size expects.
+ for row in run.dataset.rows[:PARAPHRASES_PER_FACT]:
+ print(" q:", row["messages"][0]["content"])
+
+ model = slm.load("Qwen/Qwen2.5-1.5B-Instruct")
+ result = model.finetune(run.dataset, method="more_plus",
+ more_plus_group_size=PARAPHRASES_PER_FACT)
+ print("final loss:", result.loss)
+ model.save("out/handbook_experts", fmt="adapter")
+
+ # ask it something phrased nothing like the document
+ print(model.generate("remind me how the expense thing works?"))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/examples/synthesize_from_task.py b/examples/synthesize_from_task.py
new file mode 100644
index 0000000..adc322e
--- /dev/null
+++ b/examples/synthesize_from_task.py
@@ -0,0 +1,52 @@
+"""shadow a task you have no data for yet
+
+The cold start. You know what the model should do, but nobody has run the agent
+in production, so there is nothing to capture and no traces to read. Describe
+the task in plain English and a teacher model writes the training set — then
+train a small open model on it and own the task.
+
+Run from the repo root:
+ OPENAI_API_KEY=sk-... python examples/synthesize_from_task.py
+
+The teacher here is a frontier model over an OpenAI-compatible endpoint. Swap it
+for `slm.synth.as_teacher(slm.load("Qwen/Qwen2.5-7B-Instruct"))` to keep the
+whole loop on your own hardware — nothing else changes.
+"""
+import shadowlm as slm
+
+TASK = (
+ "Triage inbound customer emails for a SaaS billing product. Classify the "
+ "urgency (low / normal / urgent), draft a short reply in a calm support "
+ "voice, and escalate to a human whenever a refund over $200 is requested."
+)
+
+
+def main():
+ # 1. a teacher writes the data. It expands the task into distinct scenarios
+ # first, then fills each one — variety comes from the structure, not from
+ # asking nicely for it.
+ run = slm.synthesize(
+ task=TASK,
+ teacher=slm.synth.frontier("gpt-4o"),
+ n=200,
+ method="lora", # the method picks the output shape: chat rows here
+ min_score=0.7, # the teacher also judges; weak rows are dropped
+ )
+ print(run.report.summary())
+
+ # 2. every rejection is counted, so you can see what you actually got
+ for traj in run.rejected[:3]:
+ print(f" rejected ({traj.metadata['reject_reason']}): "
+ f"{traj.first_user_content()[:60]}")
+
+ # 3. train on it. run.dataset is a normal Dataset — nothing about it is
+ # special because it was synthesized.
+ run.save("out/synth_task.jsonl")
+ model = slm.load("mlx-community/Qwen2.5-0.5B-Instruct-bf16", backend="mlx")
+ result = model.finetune(run.dataset, method="lora", max_steps=60)
+ print("final loss:", result.loss, result.sparkline())
+ model.save("out/shadow_from_task", fmt="adapter")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/examples/torch/sdft.py b/examples/torch/sdft.py
new file mode 100644
index 0000000..2a2493f
--- /dev/null
+++ b/examples/torch/sdft.py
@@ -0,0 +1,23 @@
+"""sdft · torch backend
+
+SDFT — on-policy self-distillation: the model samples its own answers and is
+pulled toward the same model reading the golden response in-context (adapters
+disabled), so it learns the task with far less forgetting than SFT. Steps are
+slower than lora — each one rolls out completions.
+Run from the repo root:
+ python examples/torch/sdft.py
+"""
+import shadowlm as slm
+
+
+def main():
+ ds = slm.Dataset.from_jsonl("examples/data/chat.jsonl")
+ model = slm.load("Qwen/Qwen3-8B", backend="torch", device="cuda")
+ run = model.finetune(ds, method="sdft", max_steps=60,
+ sdft_max_completion_length=128)
+ print("final loss:", run.loss, run.sparkline())
+ model.save("out/torch_sdft", fmt="adapter")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/examples/torch/sdpo.py b/examples/torch/sdpo.py
new file mode 100644
index 0000000..cd1c18a
--- /dev/null
+++ b/examples/torch/sdpo.py
@@ -0,0 +1,35 @@
+"""sdpo · torch backend
+
+SDPO — RL via self-distillation: the feedback-conditioned model teaches itself.
+Reward fns may return (score, feedback) pairs; the feedback (and any successful
+sibling rollout) becomes the self-teacher's in-context signal.
+Run from the repo root:
+ python examples/torch/sdpo.py
+"""
+import shadowlm as slm
+
+
+prompts = [
+ {"prompt": "What port does the ShadowLM studio serve on? Answer with just the number.", "answer": "8329"},
+ {"prompt": "Which backend is ShadowLM's production training path? Answer with one word.", "answer": "torch"},
+ {"prompt": "What does the M in MoRE stand for? Answer with one word.", "answer": "mixture"},
+]
+
+# 1.0 on a hit; on a miss, (0.0, hint) — the hint reaches the self-teacher.
+def reward(prompts, completions, answer=None, types=None, **kwargs):
+ golds = answer if isinstance(answer, list) else [answer] * len(completions)
+ return [1.0 if (g or "").lower() in c.lower()
+ else (0.0, f"A correct answer contains {g!r}.")
+ for c, g in zip(completions, golds)]
+
+
+def main():
+ model = slm.load("Qwen/Qwen3-8B", backend="torch", device="cuda")
+ run = model.finetune(prompts, method="sdpo", reward_fns=[reward],
+ max_steps=60)
+ print("final loss:", run.loss, run.sparkline())
+ model.save("out/torch_sdpo", fmt="adapter")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/frontend/package-lock.json b/frontend/package-lock.json
index e6bb65b..b145bd3 100644
--- a/frontend/package-lock.json
+++ b/frontend/package-lock.json
@@ -9,6 +9,7 @@
"version": "0.0.0",
"dependencies": {
"@tailwindcss/vite": "^4.3.0",
+ "@tanstack/react-query": "^5.104.1",
"class-variance-authority": "^0.7.1",
"clsx": "^2.1.1",
"lucide-react": "^1.18.0",
@@ -3187,6 +3188,32 @@
"vite": "^5.2.0 || ^6 || ^7 || ^8"
}
},
+ "node_modules/@tanstack/query-core": {
+ "version": "5.104.1",
+ "resolved": "https://registry.npmjs.org/@tanstack/query-core/-/query-core-5.104.1.tgz",
+ "integrity": "sha512-TiRghcrGkUTE+M4JBxvpPSpBhxvIzQairw+cksrHYGr+2uwC77yW1SY9nYxnYRYOppvucf8rEg6b5byoz4rTpw==",
+ "license": "MIT",
+ "funding": {
+ "type": "github",
+ "url": "https://github.com/sponsors/tannerlinsley"
+ }
+ },
+ "node_modules/@tanstack/react-query": {
+ "version": "5.104.1",
+ "resolved": "https://registry.npmjs.org/@tanstack/react-query/-/react-query-5.104.1.tgz",
+ "integrity": "sha512-Zz70EgjahNI7aCz8BZxM3vUGU44ycxEAzMuCUjLeA7SVhhOkg3ZgPjPJPvHjs6Rmky/dtmnu1WzjGHNHmhCoZQ==",
+ "license": "MIT",
+ "dependencies": {
+ "@tanstack/query-core": "5.104.1"
+ },
+ "funding": {
+ "type": "github",
+ "url": "https://github.com/sponsors/tannerlinsley"
+ },
+ "peerDependencies": {
+ "react": "^18 || ^19"
+ }
+ },
"node_modules/@ts-morph/common": {
"version": "0.27.0",
"resolved": "https://registry.npmjs.org/@ts-morph/common/-/common-0.27.0.tgz",
diff --git a/frontend/package.json b/frontend/package.json
index 4eec543..7421b95 100644
--- a/frontend/package.json
+++ b/frontend/package.json
@@ -11,6 +11,7 @@
},
"dependencies": {
"@tailwindcss/vite": "^4.3.0",
+ "@tanstack/react-query": "^5.104.1",
"class-variance-authority": "^0.7.1",
"clsx": "^2.1.1",
"lucide-react": "^1.18.0",
diff --git a/frontend/src/api.ts b/frontend/src/api.ts
index ac9dd2c..c6ec5af 100644
--- a/frontend/src/api.ts
+++ b/frontend/src/api.ts
@@ -197,6 +197,42 @@ export const addHFDataset = (
eval_split: evalSplit || null }) });
export const deleteDataset = (id: string) =>
api<{ ok: boolean }>(`/v1/datasets/${id}`, { method: "DELETE" });
+
+// ---- synthesis: make a dataset instead of bringing one ----------------------
+export interface SynthStatus {
+ synth_id: string;
+ name: string;
+ status: "running" | "succeeded" | "failed" | "stopped";
+ kept: number;
+ requested: number;
+ tokens?: number; // what the provider billed — zero for a local teacher
+ // live phase counters — a round's generating/judging batches tick as each
+ // job lands, so the bar moves instead of waiting for the whole round
+ phase?: "starting" | "planning" | "generating" | "judging" | "kept";
+ done?: number;
+ total?: number;
+ dataset_id?: string;
+ error?: string;
+ logs?: string[];
+}
+export interface SynthRequest {
+ name: string;
+ n: number;
+ method?: string;
+ min_score?: number;
+ task?: string;
+ document?: string;
+ dataset_id?: string;
+ include_seed?: boolean; // the saved dataset also carries the dataset_id rows it was written from
+ project_id?: string; // shows on that project while it writes; the result becomes its data
+ // "frontier": the frontier model saved in settings (its stored key)
+ teacher: { kind: "openai" | "local" | "frontier"; model: string; base_url?: string; api_key?: string };
+}
+export const startSynth = (body: SynthRequest) =>
+ api<{ synth_id: string }>("/v1/synth", { method: "POST", body: JSON.stringify(body) });
+export const getSynthRun = (id: string) => api(`/v1/synth/${id}`);
+export const cancelSynth = (id: string) =>
+ api<{ ok: boolean }>(`/v1/synth/${id}/cancel`, { method: "POST" });
export const getModels = () =>
api<{ catalog: CatalogModel[]; recent: string[]; server_backend: string }>("/v1/models");
export const getDownloads = () =>
@@ -207,7 +243,14 @@ export const addCustomModel = (model: string) =>
api<{ custom: CatalogModel[] }>("/v1/models/custom", { method: "POST", body: JSON.stringify({ model }) });
export const removeCustomModel = (model: string) =>
api<{ custom: CatalogModel[] }>("/v1/models/custom", { method: "POST", body: JSON.stringify({ model, remove: true }) });
-export const getSettings = () => api<{ hf_token_set: boolean }>("/v1/settings");
+export interface FrontierInfo { base_url: string; model: string }
+export interface Settings { hf_token_set: boolean; frontier: FrontierInfo | null }
+export const getSettings = () => api("/v1/settings");
+// The user's frontier model (any OpenAI-compatible API): an evaluation baseline,
+// the judge, and the upstream an agent's captured traffic passes through. The
+// key is stored on the server and never comes back. null removes it.
+export const setFrontier = (frontier: { base_url: string; api_key: string; model: string } | null) =>
+ api("/v1/settings", { method: "POST", body: JSON.stringify({ frontier }) });
export const setHfToken = (hf_token: string) =>
api<{ hf_token_set: boolean }>("/v1/settings", { method: "POST", body: JSON.stringify({ hf_token }) });
export const getVram = () =>
@@ -262,3 +305,132 @@ export const prewarm = (model: string, adapter: string | null, checkpoint: numbe
method: "POST", body: JSON.stringify({ model, adapter, checkpoint }) });
export const chat = (body: object) =>
api<{ text: string }>("/v1/chat", { method: "POST", body: JSON.stringify(body) });
+
+// ---- chat as a background task ------------------------------------------------
+// The first answer from a cold model can take longer than the proxy in front of
+// the studio holds a request, so a chat is started, then polled until it lands.
+export interface ChatTask { task_id: string; status: "running" | "succeeded" | "failed"; text: string | null; error: string | null }
+export const startChat = (body: object) =>
+ api<{ task_id: string }>("/v1/tasks/chat", { method: "POST", body: JSON.stringify(body) });
+export const getChatTask = (id: string) => api(`/v1/tasks/${id}`);
+export async function chatAsync(body: object, { every = 1200, timeoutMs = 15 * 60_000 } = {}): Promise<{ text: string }> {
+ const { task_id } = await startChat(body);
+ const end = Date.now() + timeoutMs;
+ for (;;) {
+ const t = await getChatTask(task_id);
+ if (t.status === "succeeded") return { text: t.text ?? "" };
+ if (t.status === "failed") throw new Error(t.error || "the model couldn't answer");
+ if (Date.now() > end) throw new Error("no answer after 15 minutes");
+ await new Promise((r) => setTimeout(r, every));
+ }
+}
+
+// ---- projects: one fine-tune, as Business mode walks it -----------------------
+// goal: "knowledge" (teach it facts) · "task" (teach it a job from examples) ·
+// "takeover" (move a task off a frontier model, from captured agent traffic).
+// The stage is read off the links: no dataset → Data; a run → Fine-tune; an
+// eval → Evaluate. Deploy comes with the serving endpoint.
+export type ProjectGoal = "knowledge" | "task" | "takeover";
+export interface Project {
+ project_id: string;
+ name: string;
+ goal: ProjectGoal;
+ dataset_id: string | null;
+ run_id: string | null;
+ eval_id: string | null;
+ synth_id?: string | null; // examples being written for it (the synthesizer)
+ run_dataset_id?: string | null; // what the current version trained on
+ // earlier versions, oldest first: each fine-tune this project replaced
+ history: { run_id: string; eval_id: string | null; dataset_id: string | null; replaced: number }[];
+ created: number;
+ updated: number;
+}
+export const getProjects = () => api<{ projects: Project[] }>("/v1/projects");
+export const getProject = (id: string) => api(`/v1/projects/${id}`);
+export const createProject = (name: string, goal: ProjectGoal) =>
+ api("/v1/projects", { method: "POST", body: JSON.stringify({ name, goal }) });
+export const updateProject = (id: string, patch: Partial>) =>
+ api(`/v1/projects/${id}`, { method: "PATCH", body: JSON.stringify(patch) });
+export const deleteProject = (id: string) =>
+ api<{ ok: boolean }>(`/v1/projects/${id}`, { method: "DELETE" });
+
+// ---- evaluations: task quality, as a background job ---------------------------
+// Several targets (a base model, a fine-tune at a checkpoint) answer the same
+// questions; each gets a score and per-question detail. "contains": the
+// expected answer appears in the output · "exact": it is the output.
+export type EvalMetric = "contains" | "exact" | "judge"; // judge: the frontier model scores
+// kind "frontier" is the configured frontier model (the server fills in model)
+export interface EvalTarget { label: string; model: string; adapter?: string | null; checkpoint?: number | null; kind?: "frontier" }
+export interface EvalExample { input: string; output: string; expected: string; score: number }
+export interface EvalTargetResult { score: number; n: number; examples?: EvalExample[] }
+export interface Evaluation {
+ eval_id: string;
+ name: string;
+ dataset_id: string;
+ metric: EvalMetric;
+ targets: EvalTarget[];
+ sample: number | null;
+ status: "pending" | "running" | "succeeded" | "failed";
+ error: string | null;
+ results: EvalTargetResult[]; // in target order; fills in as each finishes
+ created: number;
+ finished: number | null;
+}
+export const getEvals = () => api<{ evals: Evaluation[] }>("/v1/evals");
+export const getEval = (id: string) => api(`/v1/evals/${id}`);
+export const startEval = (body: { dataset_id: string; targets: (EvalTarget | { kind: "frontier"; label?: string })[]; metric?: EvalMetric; sample?: number; name?: string }) =>
+ api("/v1/evals", { method: "POST", body: JSON.stringify(body) });
+export const deleteEval = (id: string) =>
+ api<{ ok: boolean }>(`/v1/evals/${id}`, { method: "DELETE" });
+
+// ---- captures: an agent's traffic to its frontier model, recorded -------------
+// Point the agent's OpenAI base URL at `${origin}/v1/capture/`; each
+// call passes through to the frontier model and is recorded, then the episodes
+// become a dataset. The id is the URL's only secret.
+export interface CaptureInfo {
+ capture_id: string;
+ name: string;
+ project_id: string | null;
+ status: "open" | "closed";
+ created: number;
+ calls: number;
+ episodes?: number;
+ preview?: { question: string; answer: string; turns: number }[];
+}
+export const captureBaseUrl = (id: string) => `${window.location.origin}/v1/capture/${id}`;
+export const getCaptures = () => api<{ captures: CaptureInfo[] }>("/v1/captures");
+export const getCapture = (id: string) => api(`/v1/captures/${id}`);
+export const createCapture = (name: string, project_id?: string | null) =>
+ api("/v1/captures", { method: "POST", body: JSON.stringify({ name, project_id: project_id ?? null }) });
+export const closeCapture = (id: string) =>
+ api<{ ok: boolean }>(`/v1/captures/${id}/close`, { method: "POST" });
+export const captureToDataset = (id: string, name: string) =>
+ api(`/v1/captures/${id}/dataset`, { method: "POST", body: JSON.stringify({ name }) });
+export const deleteCapture = (id: string) =>
+ api<{ ok: boolean }>(`/v1/captures/${id}`, { method: "DELETE" });
+
+// OpenTelemetry GenAI traces (OTLP JSON, or a list of spans) → a chat dataset.
+export const importTraces = (name: string, traces: unknown) =>
+ api("/v1/datasets/traces", { method: "POST", body: JSON.stringify({ name, traces }) });
+
+// ---- deployments: a fine-tune behind an OpenAI-compatible endpoint ------------
+// An agent uses it with base URL `${origin}/openai/v1`, the deployment's key and
+// its name as the model. The key is returned once, at creation.
+export interface Deployment {
+ deployment_id: string;
+ name: string;
+ run_id: string;
+ base_model: string;
+ checkpoint: number | null;
+ project_id: string | null;
+ key_prefix: string;
+ created: number;
+ requests: number;
+ last_used: number | null;
+}
+export const deploymentBaseUrl = () => `${window.location.origin}/openai/v1`;
+export const getDeployments = () => api<{ deployments: Deployment[] }>("/v1/deployments");
+export const createDeployment = (body: { name: string; run_id: string; checkpoint?: number | null; project_id?: string | null }) =>
+ api<{ deployment: Deployment; key: string }>("/v1/deployments", { method: "POST", body: JSON.stringify(body) });
+export const deleteDeployment = (id: string) =>
+ api<{ ok: boolean }>(`/v1/deployments/${id}`, { method: "DELETE" });
diff --git a/frontend/src/App.tsx b/frontend/src/app/App.tsx
similarity index 72%
rename from frontend/src/App.tsx
rename to frontend/src/app/App.tsx
index 8e982c2..8a2fd91 100644
--- a/frontend/src/App.tsx
+++ b/frontend/src/app/App.tsx
@@ -3,13 +3,13 @@
// hash router, and the sign-in gate. Embedded in a host console
// (lib/embed.ts), the host's menu and sign-in replace both.
import {
- ArrowRight, BookOpen, Box, Cpu, Database, ExternalLink, History, KeyRound, LayoutDashboard,
- LoaderCircle, type LucideIcon, LogOut, MessagesSquare, MonitorSmartphone, Moon,
- PanelLeftClose, PanelLeftOpen, Sun, Zap,
+ ArrowRight, BookOpen, Box, ClipboardCheck, Cloud, Rocket, Cpu, Database, ExternalLink, FlaskConical, FolderKanban,
+ History, KeyRound, LayoutDashboard, LoaderCircle, type LucideIcon, LogOut, MessagesSquare,
+ MonitorSmartphone, Moon, PanelLeftClose, PanelLeftOpen, Plus, Sun, Target, Zap,
} from "lucide-react";
import { AnimatePresence, motion, MotionConfig } from "motion/react";
import { ThemeProvider, useTheme } from "next-themes";
-import { type CSSProperties, type FormEvent, type ReactNode, useEffect, useState } from "react";
+import { type CSSProperties, type FormEvent, type ReactNode, useEffect, useState, useSyncExternalStore } from "react";
import {
apiKey, clearVram, getAuthInfo, getHealth, getMethods, getSettings, getVram,
@@ -22,40 +22,52 @@ import {
} from "@/components/ui/dialog";
import { Input } from "@/components/ui/input";
import { Label } from "@/components/ui/label";
+import { FrontierDialog, useFrontier } from "@/components/frontier-settings";
+import { ModeChooser } from "@/components/mode-chooser";
import {
allows, embedded, hashToPage, type MenuItem, onHostAction, reportMenu, reportRoute, reportTitle, useEmbedTheme,
} from "@/lib/embed";
+import { useRoute } from "@/app/router";
+import Cockpit from "@/features/cockpit/Cockpit";
+import CockpitStart from "@/features/cockpit/CockpitStart";
+import { type Mode, setMode, useMode } from "@/lib/mode";
import { cn } from "@/lib/utils";
-import Dashboard from "@/pages/Dashboard";
-import Datasets from "@/pages/Datasets";
-import Machines from "@/pages/Machines";
-import Models from "@/pages/Models";
-import Playground from "@/pages/Playground";
-import Runs from "@/pages/Runs";
-import Train from "@/pages/Train";
-
-function useHash(): string {
- const [h, setH] = useState(window.location.hash);
- useEffect(() => {
- const f = () => setH(window.location.hash);
- window.addEventListener("hashchange", f);
- return () => window.removeEventListener("hashchange", f);
- }, []);
- return h.replace(/^#/, "");
-}
+import Dashboard from "@/features/overview/Dashboard";
+import Datasets from "@/features/datasets/Datasets";
+import Deployments from "@/features/deployments/Deployments";
+import Evaluate from "@/features/evaluate/Evaluate";
+import Machines from "@/features/machines/Machines";
+import Models from "@/features/models/Models";
+import Playground from "@/features/playground/Playground";
+import Projects from "@/features/projects/Projects";
+import Runs from "@/features/runs/Runs";
+import Train from "@/features/train/Train";
+
+// The current page, from the router (app/router.ts).
+const useHash = (): string => useRoute().hash;
interface NavItem { hash: string; label: string; icon: LucideIcon }
type Section = { title?: string; items: NavItem[] };
-// The navigation leads with the Playground, where you talk to what you own,
-// then follows the shadowing loop: bring data and a base model, train, watch
-// the run. Machines, where training runs, is setup, so it sits apart at the
-// foot.
-const sections: Section[] = [
+// Business mode walks one job at a time: its projects, a new one, the
+// playground. Research mode leads with the Playground, then the fine-tuning
+// loop's objects: data and base models, runs, evaluations. Machines, where
+// training runs, is setup, so it sits apart at the foot (Research only).
+const businessSections: Section[] = [
+ {
+ items: [
+ { hash: "", label: "Projects", icon: FolderKanban },
+ { hash: "projects/new", label: "New project", icon: Plus },
+ { hash: "playground", label: "Playground", icon: MessagesSquare },
+ ],
+ },
+];
+const researchSections: Section[] = [
{
items: [
{ hash: "playground", label: "Playground", icon: MessagesSquare },
{ hash: "", label: "Overview", icon: LayoutDashboard },
+ { hash: "projects", label: "Projects", icon: FolderKanban },
],
},
{
@@ -70,17 +82,30 @@ const sections: Section[] = [
items: [
{ hash: "train", label: "New run", icon: Cpu },
{ hash: "runs", label: "Runs", icon: History },
+ { hash: "evaluate", label: "Evaluate", icon: ClipboardCheck },
+ { hash: "deployments", label: "Deployments", icon: Rocket },
],
},
];
+const sectionsFor = (m: Mode | null) => (m === "research" ? researchSections : businessSections);
const machinesItem: NavItem = { hash: "machines", label: "Machines", icon: MonitorSmartphone };
-const allItems = [...sections.flatMap((s) => s.items), machinesItem];
+const allItems = [...businessSections, ...researchSections].flatMap((s) => s.items).concat(machinesItem);
+
+// The nav item a hash belongs to: "projects/" sits under Projects, a run
+// under Runs.
+function activeHash(m: Mode | null, hash: string): string {
+ const [section, arg] = hash.split("/");
+ if (section === "projects") return arg === "new" && m === "business" ? "projects/new" : m === "research" ? "projects" : "";
+ if (m === "business" && section === "") return "";
+ return section;
+}
// glyphs are the menu's icons by name, for a host that draws the menu itself
// (HostMenu): lucide's names, which the host knows.
const glyphs = new Map([
[LayoutDashboard, "layout-dashboard"], [MessagesSquare, "messages-square"], [Database, "database"],
[Box, "box"], [Cpu, "cpu"], [History, "history"], [MonitorSmartphone, "monitor-smartphone"],
+ [FolderKanban, "folder-kanban"], [Plus, "plus"], [ClipboardCheck, "clipboard-check"], [Rocket, "rocket"],
]);
// The repository; its README is the documentation.
@@ -94,6 +119,20 @@ const pageRight = 32;
const spring = { type: "spring", stiffness: 380, damping: 36 } as const;
const storageKey = "of-island-open";
+// Phone widths: the island stays a rail, and opening it lays it over the page
+// instead of pushing the page into a sliver.
+const narrowQuery = "(max-width: 767px)";
+function useNarrow(): boolean {
+ return useSyncExternalStore(
+ (l) => {
+ const m = window.matchMedia(narrowQuery);
+ m.addEventListener("change", l);
+ return () => m.removeEventListener("change", l);
+ },
+ () => window.matchMedia(narrowQuery).matches,
+ );
+}
+
function readOpen(): boolean {
try {
return localStorage.getItem(storageKey) !== "false";
@@ -236,6 +275,10 @@ function Studio({ onSignOut, embeddedIn }: { onSignOut?: () => void; embeddedIn?
const [section, arg] = hash.split("/");
const [methods, setMethods] = useState([]);
const [open, setOpen] = useState(readOpen);
+ const narrow = useNarrow();
+ const [overlay, setOverlay] = useState(false); // the island opened over the page, on a phone
+ const mode = useMode();
+ useEffect(() => { setOverlay(false); }, [hash, narrow]); // going somewhere closes it
useEffect(() => {
getMethods().then((m) => setMethods(m.methods)).catch(() => {});
@@ -246,11 +289,10 @@ function Studio({ onSignOut, embeddedIn }: { onSignOut?: () => void; embeddedIn?
useEffect(() => {
if (!embeddedIn) return;
reportRoute(hashToPage(hash));
- reportTitle(allItems.find((i) => i.hash === section)?.label ?? "Overview");
- }, [embeddedIn, hash, section]);
+ reportTitle(allItems.find((i) => i.hash === activeHash(mode, hash))?.label ?? "Projects");
+ }, [embeddedIn, hash, mode]);
- const toggle = () =>
- setOpen((o) => {
+ const toggle = () => narrow ? setOverlay((o) => !o) : setOpen((o) => {
try {
localStorage.setItem(storageKey, String(!o));
} catch {
@@ -258,8 +300,17 @@ function Studio({ onSignOut, embeddedIn }: { onSignOut?: () => void; embeddedIn?
}
return !o;
});
+ const islandOpen = narrow ? overlay : open;
+ const reserved = inset + (narrow ? railWidth : open ? openWidth : railWidth) + (narrow ? inset : gap);
const page =
+ mode === null ? :
+ section === "projects" && arg === "new" ? :
+ section === "projects" && arg ? :
+ section === "projects" ? :
+ section === "evaluate" ? :
+ section === "deployments" ? :
+ section === "" && mode === "business" ? :
section === "models" ? :
section === "datasets" ? :
section === "train" ? :
@@ -272,18 +323,22 @@ function Studio({ onSignOut, embeddedIn }: { onSignOut?: () => void; embeddedIn?
return (
{page}
-
+
);
}
return (
-
-
+
+ {narrow && overlay && (
+
);
}) : (
- No shadows yet. Train one to see it here.
+ No fine-tunes yet. Train one to see it here.
)}
@@ -274,11 +276,11 @@ export default function Playground() {
- {adapter ? "Does it cast the same shadow?" : "Talk to a model"}
+ {adapter ? "Does it beat its base?" : "Talk to a model"}
))}
- {/* until the shadow answers, both panes wait here; after, its row
+ {/* until the fine-tune answers, both panes wait here; after, its row
carries the base's wait, so don't draw a second one */}
{busy && msgs[msgs.length - 1]?.role === "user" && (
-
+
)}
) : (
@@ -318,7 +320,7 @@ export default function Playground() {
onKeyDown={(e) => { if (e.key === "Enter" && !e.shiftKey) { e.preventDefault(); send(); } }}
placeholder={!allows("operator") ? needs("operator")
: warming ? "Warming up the model — first load can take a couple of minutes…"
- : "Say something to the shadow…"}
+ : "Ask something…"}
disabled={!allows("operator")}
className="field-sizing-content max-h-40 min-h-9 min-w-0 grow resize-none rounded-lg border border-input bg-transparent px-3 py-2 text-[13px] outline-none placeholder:text-muted-foreground focus-visible:border-ring focus-visible:ring-3 focus-visible:ring-ring/50" />
diff --git a/frontend/src/features/projects/Projects.tsx b/frontend/src/features/projects/Projects.tsx
new file mode 100644
index 0000000..45c81bc
--- /dev/null
+++ b/frontend/src/features/projects/Projects.tsx
@@ -0,0 +1,192 @@
+// Projects — your models, one per job: where each stands on the loop (data,
+// fine-tune, evaluate, deploy) and its latest proof. A row opens its cockpit.
+import { useEffect, useState } from "react";
+import { ArrowRight, FolderKanban, Plus, Trash2 } from "lucide-react";
+
+import { deleteProject, getDeployments, getEval, getJobs, getProjects } from "@/api";
+import type { Evaluation, JobSummary, Project } from "@/api";
+import { ConfirmDelete, EmptyState, ErrorState, PageHeader } from "@/components/common";
+import { verdict } from "@/components/scorecard";
+import { Badge } from "@/components/ui/badge";
+import { Button } from "@/components/ui/button";
+import { Skeleton } from "@/components/ui/skeleton";
+import {
+ Table, TableBody, TableCell, TableHead, TableHeader, TableRow,
+} from "@/components/ui/table";
+import { allows } from "@/lib/embed";
+import { goalLabel } from "@/lib/recipe";
+import { cn } from "@/lib/utils";
+
+type Tone = "muted" | "primary" | "good" | "warning" | "destructive";
+
+// Where a project stands on the loop, in the cockpit's station vocabulary,
+// read off what it links to.
+export function projectStage(p: Project, run?: JobSummary, ev?: Evaluation, live?: boolean): { label: string; tone: Tone } {
+ if (!p.dataset_id) return { label: "Data · needs examples", tone: "muted" };
+ if (!p.run_id) return { label: "Fine-tune · ready", tone: "muted" };
+ if (!run || run.status === "pending" || run.status === "running") return { label: "Fine-tune · running", tone: "primary" };
+ if (run.status === "failed") return { label: "Fine-tune · failed", tone: "destructive" };
+ if (run.status === "stopped") return { label: "Fine-tune · stopped", tone: "muted" };
+ if (live) return { label: "Deploy · live", tone: "good" };
+ if (!p.eval_id) return { label: "Evaluate · ready", tone: "warning" };
+ if (!ev || ev.status === "pending" || ev.status === "running") return { label: "Evaluate · running", tone: "primary" };
+ if (ev.status === "failed") return { label: "Evaluate · failed", tone: "destructive" };
+ const v = verdict(ev);
+ return v ? { label: v.title, tone: v.tone } : { label: "Evaluated", tone: "muted" };
+}
+
+const toneClass: Record = {
+ muted: "text-muted-foreground",
+ primary: "border-primary/25 bg-primary/10 text-primary",
+ good: "border-good/30 bg-good/10 text-good",
+ warning: "border-warning/30 bg-warning/10 text-warning",
+ destructive: "border-destructive/25 bg-destructive/10 text-destructive",
+};
+
+export function StageBadge({ stage }: { stage: { label: string; tone: Tone } }) {
+ return {stage.label};
+}
+
+// The latest proof, in one line: the fine-tune's right answers against the base's.
+function proof(ev?: Evaluation): string | null {
+ if (!ev || ev.status !== "succeeded" || !ev.results[0]) return null;
+ const [t, b] = ev.results;
+ const right = (s: number, n: number) => Math.round(s * n);
+ return `Answers ${right(t.score, t.n)} of ${t.n} correctly${b ? `; base gets ${right(b.score, b.n)}` : ""}`;
+}
+
+export default function Projects() {
+ const [projects, setProjects] = useState(null);
+ const [runs, setRuns] = useState>({});
+ const [evals, setEvals] = useState>({});
+ const [liveRuns, setLiveRuns] = useState>(new Set());
+ const [err, setErr] = useState(null);
+ const [doomed, setDoomed] = useState(null);
+ const operator = allows("operator");
+
+ useEffect(() => {
+ let live = true;
+ const tick = async () => {
+ try {
+ const [{ projects }, { jobs }, deps] = await Promise.all([
+ getProjects(), getJobs(), getDeployments().catch(() => ({ deployments: [] })),
+ ]);
+ if (!live) return;
+ setProjects(projects);
+ setRuns(Object.fromEntries(jobs.map((j) => [j.job_id, j])));
+ setLiveRuns(new Set(deps.deployments.map((d) => d.run_id)));
+ setErr(null);
+ // Only the evaluations projects link to, and only until they finish.
+ const want = projects.map((p) => p.eval_id).filter((id): id is string => !!id);
+ const fetched = await Promise.all(want.map((id) => getEval(id).catch(() => null)));
+ if (!live) return;
+ setEvals(Object.fromEntries(fetched.filter((e): e is Evaluation => !!e).map((e) => [e.eval_id, e])));
+ } catch (e) {
+ if (live) setErr(e);
+ }
+ };
+ tick();
+ const t = setInterval(tick, 4000);
+ return () => { live = false; clearInterval(t); };
+ }, []);
+
+ async function remove(p: Project) {
+ setDoomed(null);
+ try {
+ await deleteProject(p.project_id);
+ setProjects((ps) => ps?.filter((x) => x.project_id !== p.project_id) ?? ps);
+ } catch (e) {
+ setErr(e);
+ }
+ }
+
+ const newButton = operator && (
+ New model
+ );
+
+ return (
+ <>
+
+
+ {err != null && projects === null && }
+
+ {projects === null && err == null && (
+
+ {[0, 1, 2].map((i) => )}
+
+ )}
+
+ {projects?.length === 0 && (
+
+
+ Tell Ctrl agent what your model should do and give it examples. It picks the base model and settings,
+ fine-tunes, proves the result on your own questions against the base model, and deploys it when it's
+ ready, showing you every choice along the way.
+
+ {operator && (
+ Make your first model
+ )}
+