From d5a1a4d4e377188a1fa703ee462ade3f17db113e Mon Sep 17 00:00:00 2001
From: Khush Patel
Date: Fri, 4 Sep 2026 10:42:05 -0700
Subject: [PATCH 1/6] Fix the Apple Silicon install, and read trace rewards
from any span
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Two bugs, found while wiring an evaluator's scored traces into training.
`pip install shadowlm` could not resolve on an arm64 Mac at all. mlx-lm-lora
was a base dependency, every published version requires mlx_lm>=0.30.6, and
mlx-lm 0.30+ pins transformers==5.0.0rc — which the torch trainer's <4.57
ceiling cannot admit. pip reported it as an unsolvable base install rather
than as the version clash it is.
mlx-lm-lora moves to the `preference` extra, which is where the code already
said it lived: both call sites import it lazily and the ImportError already
told people to run `pip install shadowlm[preference]`. mlx-lm is capped below
0.30 so the base install stays inside the transformers window — 0.29.x is the
last line taking transformers>=4.39.3 unbounded.
The extra still cannot share an environment with the torch path, because that
conflict is upstream and real. Rather than leave that as a puzzle, the
ImportError now says so and gives the two commands that work. A base install
resolves clean on Apple Silicon: mlx-lm 0.29.1, transformers 4.56.2, torch
2.14.0, trl 0.24.0.
Second: `from_spans` read `reward_key` only from spans that parsed as a model
call, because the read sat below the `if call is None: continue`. A reward
describes the episode, not one call, so an evaluator that scores a whole trace
writes it on the root — and the root is not a model call. The reward was
dropped and the trajectory came back at 0.0, which is indistinguishable from a
genuinely bad episode. Silent, and it poisons anything downstream that filters
on min_reward or ranks a preference pair.
The read moves above the filter, and both the reward reader and the call
builder now derive the episode id through one `_trace_of` — they key the same
dict, so a second copy would break on the first span carrying
gen_ai.conversation.id.
Four tests cover it, including that conversation id and that an unparseable
reward is ignored rather than fatal. Two of them fail on the previous commit.
218 tests pass.
Co-Authored-By: Claude Opus 5
Claude-Session: https://claude.ai/code/session_017t48BoeNVguFPXaARs8SHP
---
dev.py | 17 ++++++++++++
pyproject.toml | 30 ++++++++++++++++++---
shadowlm/backends/mlx.py | 15 ++++++++---
shadowlm/traces.py | 46 ++++++++++++++++++++++++-------
tests/test_traces.py | 58 ++++++++++++++++++++++++++++++++++++++++
5 files changed, 149 insertions(+), 17 deletions(-)
create mode 100644 dev.py
diff --git a/dev.py b/dev.py
new file mode 100644
index 0000000..a92038a
--- /dev/null
+++ b/dev.py
@@ -0,0 +1,17 @@
+from lyzr import Studio
+
+# Initialize the SDK
+studio = Studio(api_key="your-api-key")
+
+# Create an agent
+agent = studio.create_agent(
+ name="My Assistant",
+ provider="gpt-4o",
+ role="Helpful assistant",
+ goal="Help users with their questions",
+ instructions="Be concise and accurate"
+)
+
+# Run the agent
+response = agent.run("What is machine learning?")
+print(response.response)
\ No newline at end of file
diff --git a/pyproject.toml b/pyproject.toml
index 2e4a5d8..38c3700 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -62,9 +62,13 @@ dependencies = [
"typer>=0.12",
"rich>=13",
"pyyaml>=6",
- # Apple-Silicon dev loop — installed automatically on arm64 macOS only
- "mlx-lm>=0.20; sys_platform == 'darwin' and platform_machine == 'arm64'",
- "mlx-lm-lora>=2.0; sys_platform == 'darwin' and platform_machine == 'arm64'",
+ # Apple-Silicon dev loop — installed automatically on arm64 macOS only.
+ #
+ # Capped below 0.30 on purpose: mlx-lm 0.30+ pins `transformers==5.0.0rc*`,
+ # which cannot coexist with the <4.57 ceiling above, and pip reports that
+ # as an unsolvable base install rather than as the version clash it is.
+ # 0.29.x is the last line that takes `transformers>=4.39.3` unbounded.
+ "mlx-lm>=0.20,<0.30; sys_platform == 'darwin' and platform_machine == 'arm64'",
]
[project.urls]
@@ -90,13 +94,31 @@ verl = [
dev = [
"pytest>=8",
]
+# Preference training on Apple Silicon (DPO/GRPO via the mlx backend).
+#
+# Separate because it cannot share an environment with the torch path: every
+# published mlx-lm-lora requires `mlx_lm>=0.30.6`, which requires
+# `transformers==5.0.0rc*`, and the torch trainer needs `transformers<4.57`
+# (see the ceiling above). Installing this into the base environment makes the
+# whole install unsolvable, which is why it is not a base dependency.
+#
+# Use it in an environment of its own:
+#
+# pip install shadowlm[preference] --no-deps
+# pip install "mlx-lm-lora>=2.0"
+#
+# Everything else — LoRA, QLoRA, DoRA, CPT, SFT from traces — works in the base
+# install with no extra.
+preference = [
+ "mlx-lm-lora>=2.0; sys_platform == 'darwin' and platform_machine == 'arm64'",
+]
+
# Back-compat aliases — the training stack now ships in the base install, so
# these resolve to nothing but keep older `shadowlm[all]` / `[torch]` commands working.
all = []
mlx-all = []
torch = []
mlx = []
-preference = []
retrieval = []
cli = []
diff --git a/shadowlm/backends/mlx.py b/shadowlm/backends/mlx.py
index ef39c33..332c180 100644
--- a/shadowlm/backends/mlx.py
+++ b/shadowlm/backends/mlx.py
@@ -792,8 +792,12 @@ def _finetune_dpo(self, dataset: Dataset, config: TrainConfig, callbacks: Callba
from mlx_lm_lora.trainer.dpo_trainer import DPOTrainingArgs, train_dpo # noqa: PLC0415
except ImportError as e:
raise ImportError(
- "Preference training on Apple Silicon needs mlx-lm-lora: "
- "pip install shadowlm[preference]"
+ "Preference training on Apple Silicon needs mlx-lm-lora, which requires "
+ "mlx-lm 0.30+ and transformers 5 — versions the torch trainer "
+ "cannot share an environment with. Install it in one of its own:\n\n"
+ " pip install shadowlm[preference] --no-deps\n"
+ " pip install 'mlx-lm-lora>=2.0'\n\n"
+ "LoRA, QLoRA, DoRA and CPT need none of this."
) from e
from mlx_lm import load as mlx_load # noqa: PLC0415
from mlx_lm.tuner.utils import linear_to_lora_layers # noqa: PLC0415
@@ -873,7 +877,12 @@ def _finetune_grpo(self, dataset: Dataset, config: TrainConfig, callbacks: Callb
from mlx_lm_lora.trainer.grpo_trainer import GRPOTrainingArgs, train_grpo # noqa: PLC0415
except ImportError as e:
raise ImportError(
- "GRPO on Apple Silicon needs mlx-lm-lora: pip install shadowlm[preference]"
+ "GRPO on Apple Silicon needs mlx-lm-lora, which requires "
+ "mlx-lm 0.30+ and transformers 5 — versions the torch trainer "
+ "cannot share an environment with. Install it in one of its own:\n\n"
+ " pip install shadowlm[preference] --no-deps\n"
+ " pip install 'mlx-lm-lora>=2.0'\n\n"
+ "LoRA, QLoRA, DoRA and CPT need none of this."
) from e
from mlx_lm import load as mlx_load # noqa: PLC0415
from mlx_lm.tuner.utils import linear_to_lora_layers # noqa: PLC0415
diff --git a/shadowlm/traces.py b/shadowlm/traces.py
index 0e5ec38..e745db5 100644
--- a/shadowlm/traces.py
+++ b/shadowlm/traces.py
@@ -335,6 +335,22 @@ def __init__(self, trace, ts, messages, response, tools, model):
self.tools, self.model = tools, model
+def _trace_of(span: dict, attrs: dict) -> str:
+ """The episode a span belongs to.
+
+ `gen_ai.conversation.id` wins where it is set: one conversation can span
+ several OTel traces, and the episode is the conversation. Shared by the
+ call builder and the reward reader so both key the same episode.
+ """
+ return str(
+ attrs.get(_CONVERSATION)
+ or span.get("trace_id")
+ or span.get("traceId")
+ or span.get("span_id")
+ or ""
+ )
+
+
def _span_call(span: dict) -> _Call | None:
"""An LLM/chat span → a (prompt, response) call, or None if it isn't one.
@@ -354,8 +370,7 @@ def _span_call(span: dict) -> _Call | None:
if sysmsgs and not (prompt and prompt[0].get("role") == "system"):
prompt = sysmsgs + prompt
model = attrs.get(_RESP_MODEL) or attrs.get(_REQ_MODEL) or attrs.get("llm.model_name")
- trace = (attrs.get(_CONVERSATION) or span.get("trace_id") or span.get("traceId")
- or span.get("span_id") or "")
+ trace = _trace_of(span, attrs)
ts = span.get("start_time") or span.get("startTimeUnixNano") or 0
try:
ts = float(ts)
@@ -410,7 +425,9 @@ def from_spans(
and, for grouping, a `trace_id`. `builder` is "conversation" (default — fold
each trace's agent loop into one multi-turn episode) or "per_request" (one
episode per model call). `reward_key`, if given, reads a per-trace scalar
- from that span attribute and sets it as the trajectory `reward`.
+ from that span attribute and sets it as the trajectory `reward` — from any
+ span in the trace, including the root, since a reward describes the
+ episode rather than one model call.
"""
if builder not in ("conversation", "per_request"):
raise ValueError(f"unknown builder {builder!r} (conversation | per_request)")
@@ -418,17 +435,26 @@ def from_spans(
by_trace: dict[str, list[_Call]] = {}
rewards: dict[str, float] = {}
for span in spans:
+ # The reward is read before the span is filtered, because it describes
+ # the episode rather than the model call. An evaluator that scores a
+ # whole trace naturally writes it on the root — and the root is not a
+ # model call, so reading it only from calls silently dropped it and
+ # left the trajectory at 0.0. A zero reward is indistinguishable from
+ # a bad episode, so that loss was invisible.
+ if reward_key is not None:
+ attrs = _flatten(span.get("attributes", {}))
+ if reward_key in attrs:
+ trace_id = _trace_of(span, attrs)
+ if trace_id:
+ try:
+ rewards[trace_id] = float(attrs[reward_key])
+ except (TypeError, ValueError):
+ pass
+
call = _span_call(span)
if call is None:
continue
by_trace.setdefault(call.trace, []).append(call)
- if reward_key is not None:
- attrs = _flatten(span.get("attributes", {}))
- if reward_key in attrs:
- try:
- rewards[call.trace] = float(attrs[reward_key])
- except (TypeError, ValueError):
- pass
out: list[Trajectory] = []
for trace, calls in by_trace.items():
diff --git a/tests/test_traces.py b/tests/test_traces.py
index 3dc5a03..8fc9957 100644
--- a/tests/test_traces.py
+++ b/tests/test_traces.py
@@ -256,3 +256,61 @@ def test_sample_export_file_end_to_end():
["system", "user", "assistant", "tool", "assistant"]
ds = traces.to_dataset(eps, min_reward=0.5)
assert ds.format == "chat" and len(ds.rows) == 2
+
+
+# A reward describes the episode, not one model call. An evaluator that scores
+# a whole trace writes it on the root span — which is not a model call, so a
+# reader that filtered to calls before reading the key dropped it silently and
+# left the trajectory at 0.0. A zero reward looks exactly like a bad episode,
+# which is what made the loss invisible.
+def _root_span(trace_id, reward):
+ return {
+ "trace_id": trace_id,
+ "start_time": 0.0,
+ "name": "AgentExecutor",
+ "attributes": {"openinference.span.kind": "AGENT", "eval.score": reward},
+ }
+
+
+def test_a_reward_on_the_root_span_reaches_the_trajectory():
+ spans = [_root_span("t1", 1.0), *_agent_run("t1")]
+ # Strip the reward off the model call, so the root is the only source.
+ spans[-1]["attributes"].pop("eval.score")
+
+ trajs = traces.from_spans(spans, reward_key="eval.score")
+ assert len(trajs) == 1
+ assert trajs[0].reward == 1.0
+
+
+def test_a_root_reward_still_filters_a_dataset():
+ spans = [_root_span("t1", 0.25), *_agent_run("t1")]
+ spans[-1]["attributes"].pop("eval.score")
+
+ assert len(traces.to_dataset(spans, reward_key="eval.score", min_reward=0.2)) == 1
+ try:
+ traces.to_dataset(spans, reward_key="eval.score", min_reward=0.5)
+ except ValueError:
+ pass
+ else:
+ raise AssertionError("min_reward above the root's score should keep nothing")
+
+
+def test_a_conversation_id_keys_the_reward_the_same_way_it_keys_the_calls():
+ # gen_ai.conversation.id wins over trace_id when present, and the reward
+ # reader has to agree with the call builder or they key different episodes.
+ spans = _agent_run("t1")
+ for span in spans:
+ span["attributes"]["gen_ai.conversation.id"] = "conv-9"
+ spans[-1]["attributes"].pop("eval.score")
+ root = _root_span("t-other", 1.0)
+ root["attributes"]["gen_ai.conversation.id"] = "conv-9"
+
+ trajs = traces.from_spans([root, *spans], reward_key="eval.score")
+ assert len(trajs) == 1
+ assert trajs[0].reward == 1.0
+
+
+def test_an_unparseable_reward_is_ignored_not_fatal():
+ spans = [_root_span("t1", "not-a-number"), *_agent_run("t1")]
+ spans[-1]["attributes"].pop("eval.score")
+ assert traces.from_spans(spans, reward_key="eval.score")[0].reward == 0.0
From d6dd3cc7b62f364397fa4728fb074e2cb7fe3083 Mon Sep 17 00:00:00 2001
From: Khush Patel
Date: Mon, 5 Oct 2026 20:55:44 -0700
Subject: [PATCH 2/6] serve: let listed host consoles frame the studio
Every response sends frame-ancestors for the built-in origins (opencontroller
on studio-dev, localhost on any port) plus SHADOWLM_FRAME_ANCESTORS, and the
studio page carries the same list in a meta tag for the oc-embed/1 bridge.
---
shadowlm/serve.py | 62 ++++++++++++++++++++++++++++++++++++-
tests/test_serve_embed.py | 65 +++++++++++++++++++++++++++++++++++++++
2 files changed, 126 insertions(+), 1 deletion(-)
create mode 100644 tests/test_serve_embed.py
diff --git a/shadowlm/serve.py b/shadowlm/serve.py
index e0e69a1..d28804f 100644
--- a/shadowlm/serve.py
+++ b/shadowlm/serve.py
@@ -18,6 +18,7 @@
import argparse
import hashlib
import hmac
+import html
import io
import json
import os
@@ -26,6 +27,7 @@
import tarfile
import threading
import time
+import urllib.parse
import uuid
from dataclasses import dataclass, field
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
@@ -104,6 +106,55 @@ def record(self, addr: str, *, ok: bool) -> None:
(pip installs ship the built UI — you only see this from source.)
"""
+# ---- embedding: a host console may show the studio in a frame ---------------
+# A host console such as opencontroller shows the studio inside its own, in a
+# frame, speaking the "oc-embed/1" bridge (frontend/src/lib/embed.ts). Two
+# kinds of origin may frame it: the built-in ones (opencontroller on
+# studio-dev, and localhost on any port for development), plus whatever
+# SHADOWLM_FRAME_ANCESTORS lists (space- or comma-separated origins). Every
+# response sends that list as frame-ancestors, and the studio page carries it
+# in a meta tag so, once framed, it talks to nothing else.
+_FRAME_ANCESTORS_ENV = "SHADOWLM_FRAME_ANCESTORS"
+_BUILTIN_FRAME_ANCESTORS = (
+ "https://dev.opencontroller.sh",
+ "http://localhost:*",
+ "https://localhost:*",
+)
+
+
+def _parse_origins(raw: str) -> list[str]:
+ """Plain http(s) origins from a space/comma list; anything else (a path,
+ a query, credentials, another scheme) is dropped, so framing is never
+ opened wider than meant."""
+ out: list[str] = []
+ for f in re.split(r"[\s,]+", raw or ""):
+ if not f:
+ continue
+ u = urllib.parse.urlsplit(f)
+ if (u.scheme not in ("http", "https") or not u.netloc or "@" in u.netloc
+ or u.path not in ("", "/") or u.query or u.fragment):
+ continue
+ origin = f"{u.scheme}://{u.netloc.lower()}"
+ if origin not in out:
+ out.append(origin)
+ return out
+
+
+def frame_ancestors() -> list[str]:
+ """Origins that may frame the studio: built-ins, then the env's."""
+ out = list(_BUILTIN_FRAME_ANCESTORS)
+ for o in _parse_origins(os.environ.get(_FRAME_ANCESTORS_ENV, "")):
+ if o not in out:
+ out.append(o)
+ return out
+
+
+def _with_embed_parents(page: bytes, origins: list[str]) -> bytes:
+ """The studio page with the framing origins in a meta tag, for embed.ts."""
+ meta = ('').encode()
+ return page.replace(b"", meta + b"", 1)
+
@dataclass
class _Job:
@@ -1227,7 +1278,15 @@ def valid_bearer(self, token: str) -> bool:
def make_handler(server: Server, auth: "Auth"):
throttle = LoginThrottle()
+ ancestors = frame_ancestors()
+ frame_policy = "frame-ancestors 'self' " + " ".join(ancestors)
+
class Handler(BaseHTTPRequestHandler):
+ def end_headers(self):
+ # Only the listed host consoles may frame anything this serves.
+ self.send_header("Content-Security-Policy", frame_policy)
+ super().end_headers()
+
def log_message(self, fmt, *args): # quiet; job logs print directly
pass
@@ -1291,7 +1350,8 @@ def do_GET(self): # noqa: N802
if parts == [""]: # the React studio shell — no auth for the page;
static_index = Path(__file__).parent / "_static" / "index.html"
if static_index.exists(): # the API itself stays authed
- self._send(200, static_index.read_bytes(),
+ self._send(200, _with_embed_parents(
+ static_index.read_bytes(), ancestors),
ctype="text/html; charset=utf-8")
else: # only on an unbuilt source checkout — build it or use --dev
self._send(200, _NO_BUILD_PAGE.encode(),
diff --git a/tests/test_serve_embed.py b/tests/test_serve_embed.py
new file mode 100644
index 0000000..169f77c
--- /dev/null
+++ b/tests/test_serve_embed.py
@@ -0,0 +1,65 @@
+"""A host console (opencontroller) may frame the studio: only the listed
+origins, which every response names in frame-ancestors and the studio page
+carries in a meta tag for the oc-embed/1 bridge.
+"""
+
+from __future__ import annotations
+
+import threading
+import urllib.request
+from http.server import ThreadingHTTPServer
+
+import pytest
+
+from shadowlm import serve
+from shadowlm.serve import Auth, Server, _parse_origins, frame_ancestors, make_handler
+
+
+def test_parse_origins_keeps_plain_origins_only():
+ raw = ("https://console.example.com, http://Host:8080/ "
+ "https://x.test/path ftp://y.test https://u:p@z.test "
+ "https://q.test?a=1 javascript:alert(1) https://console.example.com")
+ assert _parse_origins(raw) == ["https://console.example.com", "http://host:8080"]
+
+
+def test_frame_ancestors_builtins_then_env(monkeypatch):
+ monkeypatch.setenv("SHADOWLM_FRAME_ANCESTORS", "https://oc.example.com http://localhost:*")
+ got = frame_ancestors()
+ assert got[:3] == ["https://dev.opencontroller.sh", "http://localhost:*", "https://localhost:*"]
+ assert got[3:] == ["https://oc.example.com"] # no duplicate of a built-in
+
+
+@pytest.fixture()
+def studio(tmp_path, monkeypatch):
+ monkeypatch.setenv("SHADOWLM_FRAME_ANCESTORS", "https://oc.example.com")
+ static = tmp_path / "pkg" / "_static"
+ static.mkdir(parents=True)
+ (static / "index.html").write_text("