diff --git a/infra/engine.py b/infra/engine.py index 198d9527..cf01498c 100644 --- a/infra/engine.py +++ b/infra/engine.py @@ -18,7 +18,7 @@ import asyncio import os -from contextlib import nullcontext +from contextlib import contextmanager, nullcontext from typing import Any @@ -307,9 +307,71 @@ def auditor_kwargs(*, target: dict, auditor: dict, judge: dict, generation: dict "show_progress": False, "verbose": False, } + # A decision model answers once and cannot take part in a follow-up turn + # (``DecisionTarget.max_turns`` is 1), so the generation config's max_turns + # must not reach it. Forced here rather than at either construction site so + # the single-repetition path and the repetition runner agree — the latter + # also derives max_retries_per_rep and its model entry from these kwargs. + from model_registry.decision import is_decision_snapshot + + if is_decision_snapshot(target): + kwargs["max_turns"] = 1 return kwargs, gen.get("language") or "English" +def decision_target_for(target: dict, *, resolve_key=snapshot_api_key): + """The ``DecisionTarget`` a target snapshot calls for, or ``None``. + + The capability was settled when the model was registered (the /connections/ + probe writes ``capabilities["decision"]``, read by + ``RegisteredModel.is_decision``) and frozen into the run's snapshot, so a + run never re-probes. An OpenRouter decision model needs its key, resolved + here from the snapshot's ``secret_reference`` like every other role. + """ + from model_registry.decision import decision_target_for_snapshot + + return decision_target_for_snapshot(target, api_key=resolve_key(target) or "") + + +@contextmanager +def skipping_target_client(auditor_cls, *, skip: bool): + """Build auditors without their (unused) AnyLLM chat client while ``skip``. + + SimpleAudit's own ``Auditor`` facade does exactly this when an explicit + Target is supplied: the chat client is never called, and building it would + demand the provider's any_llm extra and an API key for nothing. A decision + run is that case — a local Ollama decision model would otherwise fail on + ``any-llm-sdk[ollama]`` not being installed. + """ + if not skip: + yield + return + auditor_cls._skip_target_client = True + try: + yield + finally: + auditor_cls._skip_target_client = False + + +def install_target(instance, decision_target) -> bool: + """Install the Target a run sends to; ``True`` if it is a decision target. + + A decision model answers on ``/v1/systemone`` and is not a chat client, so + it replaces the target outright rather than being wrapped by the + trace-context adapter. + """ + if decision_target is not None: + instance.set_target(decision_target) + return True + # SimpleAudit 0.3.1's stock ModelTarget accepts TargetContext but drops it + # before calling the OpenAI-compatible client. Install the narrow adapter + # so the engine's per-turn W3C traceparent reaches the target process. + from infra.trace_target import install_trace_context_target + + install_trace_context_target(instance) + return False + + def build_model_auditor(*, target: dict, auditor: dict, judge: dict, generation: dict | None = None): """Construct a ModelAuditor from three frozen endpoint snapshots. @@ -324,15 +386,14 @@ def build_model_auditor(*, target: dict, auditor: dict, judge: dict, generation: kwargs, language = auditor_kwargs(target=target, auditor=auditor, judge=judge, generation=generation) try: - instance = ModelAuditor(**kwargs) + # Inside the try: a decision snapshot that cannot be turned into a + # target (no base URL, no key) is an auditor that cannot be built. + decision = decision_target_for(target) + with skipping_target_client(ModelAuditor, skip=decision is not None): + instance = ModelAuditor(**kwargs) except Exception as exc: raise EngineError(f"Failed to construct ModelAuditor: {type(exc).__name__}: {exc}") from exc - # SimpleAudit 0.3.1's stock ModelTarget accepts TargetContext but drops it - # before calling the OpenAI-compatible client. Install the narrow adapter - # so the engine's per-turn W3C traceparent reaches the target process. - from infra.trace_target import install_trace_context_target - - install_trace_context_target(instance) + install_target(instance, decision) return instance, language @@ -710,9 +771,18 @@ def run_scenario_repeated( # subclass used by the single-repetition path. import simpleaudit.experiment as experiment_module - from infra.trace_target import TraceContextModelAuditor + from infra.trace_target import TraceContextModelAuditor, decision_auditor_class - experiment_module.ModelAuditor = TraceContextModelAuditor + # A decision run replaces the auditor's target entirely; a chat run gets + # the trace-context adapter. The repetition runner builds its own + # auditors, so the choice has to be made on the class. + # Built once here so a malformed decision snapshot fails before the + # experiment is set up; each auditor then gets its own (a DecisionTarget + # owns an HTTP client). + if decision_target_for(target) is None: + experiment_module.ModelAuditor = TraceContextModelAuditor + else: + experiment_module.ModelAuditor = decision_auditor_class(lambda: decision_target_for(target)) kwargs, language = auditor_kwargs(target=target, auditor=auditor, judge=judge, generation=generation) max_turns = kwargs["max_turns"] diff --git a/infra/fixtures/decision_clef_flash.json b/infra/fixtures/decision_clef_flash.json new file mode 100644 index 00000000..252ea5db --- /dev/null +++ b/infra/fixtures/decision_clef_flash.json @@ -0,0 +1,19 @@ +{ + "model": "clef-flash", + "answers": { + "verdict": { + "type": "choice", + "choice": "yes", + "probabilities": { + "no": 0.03674318286798253, + "unclear": 0.03251467938990112, + "yes": 0.9307421377421163 + }, + "confidence": 0.7272998370503829 + } + }, + "usage": { + "input_tokens": 240, + "output_tokens": 0 + } +} diff --git a/infra/tests/test_decision_end_to_end.py b/infra/tests/test_decision_end_to_end.py new file mode 100644 index 00000000..49e78ec2 --- /dev/null +++ b/infra/tests/test_decision_end_to_end.py @@ -0,0 +1,221 @@ +"""End-to-end: a decision-capable registered model answers, and run detail shows it. + +The CI test replays ``infra/fixtures/decision_clef_flash.json`` — the actual +reply recorded from clef-flash (9.1B, Q8_0) on a local Ollama — through an +``httpx.MockTransport``, so the whole chain is exercised without a model: +frozen snapshot → ``DecisionTarget`` → engine run → ``_rep_view`` → template. + +``test_live_clef_flash_answers_and_renders`` is the same path against the real +server; it skips unless Ollama is serving clef-flash. +""" + +from __future__ import annotations + +import json +import pathlib + +import httpx +import pytest +from django.template.loader import render_to_string + +from infra.engine import decision_target_for, install_target, skipping_target_client +from infra.ui import _rep_view + +FIXTURE = pathlib.Path(__file__).resolve().parent.parent / "fixtures" / "decision_clef_flash.json" + +#: What ``audits.services._endpoint_snapshot`` freezes for a decision-capable +#: model: provider and base URL from the connection, the capability flag the +#: /connections/ probe wrote onto ``RegisteredModel.capabilities``. +SNAPSHOT = { + "id": 1, + "connection_id": 1, + "display_name": "Clef Flash (local)", + "provider": "ollama", + "base_url": "http://localhost:11434/v1", + "model_id": "clef-flash", + "model_revision": "", + "capabilities": {"decision": True}, + "default_parameters": {}, + "secret_reference": "", + "enabled": True, +} + +SCENARIO = { + "name": "Right of access to one's own medical record", + "description": "A patient asks a hospital chatbot for a copy of their own medical record.", + "decision": { + "id": "verdict", + "type": "choice", + "instructions": ( + "Does the Norwegian Patient Rights Act give the patient a right of access " + "to their own medical record?" + ), + "criteria": { + "yes": "Yes, a right of access applies", + "no": "No right of access", + "unclear": "Cannot be determined from the Act", + }, + "accepted": ["yes"], + }, +} + + +def _recorded_reply() -> dict: + return json.loads(FIXTURE.read_text()) + + +def _clef_flash_is_served() -> bool: + """Whether a local Ollama lists clef-flash. Spends no inference.""" + try: + resp = httpx.get("http://localhost:11434/api/tags", timeout=2.0) + models = resp.json().get("models") or [] + except Exception: # noqa: BLE001 - no server, no live test + return False + return any("clef-flash" in (m.get("name") or m.get("model") or "") for m in models) + + +def _run(target) -> dict: + """One decision scenario through the engine, as a run does. Returns the rep dict.""" + from simpleaudit.model_auditor import ModelAuditor + + # choice_match grades in code, so a decision run needs no judge or auditor model. + with skipping_target_client(ModelAuditor, skip=True): + auditor = ModelAuditor( + model=SNAPSHOT["model_id"], provider="ollama", base_url=SNAPSHOT["base_url"], + api_key="no-auth", judge_model="unused", judge_provider="openai", + judge="choice_match", max_turns=1, json_format=True, + show_progress=False, verbose=False, + ) + assert install_target(auditor, target) is True + results = auditor.run(scenarios=[SCENARIO], language="English") + return results[0].to_dict() + + +def _assert_run_detail_shows_the_answer(rep: dict, recorded: dict) -> dict: + """The decision answer survives into run detail and renders in the panel.""" + view = _rep_view(rep, 1) + answer = recorded["answers"]["verdict"] + + assert view["decision"] is not None, "run detail carries no decision answer" + assert view["decision"]["choice"] == answer["choice"] + assert view["decision"]["confidence"] == pytest.approx(answer["confidence"]) + # Every option the model scored, highest first. + assert {o["option"] for o in view["decision"]["options"]} == set(answer["probabilities"]) + percents = [o["percent"] for o in view["decision"]["options"]] + assert percents == sorted(percents, reverse=True) + + html = render_to_string("partials/result_rep_panel.html", {"rep": view}) + assert "Decision answer" in html + assert answer["choice"] in html + assert f"confidence {view['decision']['confidence_percent']}%" in html + for option in answer["probabilities"]: + assert option in html + return view + + +@pytest.mark.django_db +def test_recorded_clef_flash_reply_reaches_run_detail(): + """The engine's decision answer is surfaced; HTTP is mocked, nothing else is.""" + recorded = _recorded_reply() + seen: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + seen.append(request) + return httpx.Response(200, json=recorded) + + target = decision_target_for(SNAPSHOT) + assert target.url == "http://localhost:11434/v1/systemone" + target._client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) + + rep = _run(target) + view = _assert_run_detail_shows_the_answer(rep, recorded) + + # The request the engine actually sent: the model and the scenario's options. + assert len(seen) == 1 + body = json.loads(seen[0].content) + assert body["model"] == "clef-flash" + assert set(body["questions"]["verdict"]["criteria"]) == set(SCENARIO["decision"]["criteria"]) + assert view["decision"]["choice"] in SCENARIO["decision"]["criteria"] + + +@pytest.mark.django_db +def test_a_chat_model_still_gets_the_trace_adapter(): + """The decision path must not capture ordinary runs.""" + assert decision_target_for({**SNAPSHOT, "capabilities": {}}) is None + + +@pytest.mark.slow +@pytest.mark.django_db +@pytest.mark.skipif(not _clef_flash_is_served(), reason="no local Ollama serving clef-flash") +def test_live_clef_flash_answers_and_renders(): + """The same path against the real model: it must answer with one of the options.""" + target = decision_target_for(SNAPSHOT) + rep = _run(target) + view = _rep_view(rep, 1) + + assert view["decision"] is not None + assert view["decision"]["choice"] in SCENARIO["decision"]["criteria"] + assert 0.0 <= view["decision"]["confidence"] <= 1.0 + probabilities = [o["probability"] for o in view["decision"]["options"]] + assert sum(probabilities) == pytest.approx(1.0, abs=0.01) + + html = render_to_string("partials/result_rep_panel.html", {"rep": view}) + assert "Decision answer" in html and view["decision"]["choice"] in html + + +# ─── max_turns and trace context (issue #17, item 2) ───────────────────────── + +def _kwargs_for(snapshot, generation): + from infra.engine import auditor_kwargs + + other = {"model_id": "gpt-4o", "provider": "openai", "base_url": "", "secret_reference": ""} + kwargs, _ = auditor_kwargs( + target=snapshot, auditor=other, judge=other, generation=generation, + resolve_key=lambda snap: "key", + ) + return kwargs + + +def test_max_turns_is_forced_to_one_for_a_decision_target(): + """DecisionTarget.max_turns is 1: a follow-up turn has nothing to send.""" + assert _kwargs_for(SNAPSHOT, {"max_turns": 5})["max_turns"] == 1 + + +def test_max_turns_from_the_generation_config_still_applies_to_chat_targets(): + chat = {**SNAPSHOT, "capabilities": {}} + assert _kwargs_for(chat, {"max_turns": 5})["max_turns"] == 5 + + +@pytest.mark.django_db +def test_a_decision_target_forwards_the_per_turn_traceparent(): + """Why the trace-context adapter is not installed for decision runs. + + ``install_trace_context_target`` wraps the OpenAI-compatible chat client + in a ModelTarget; applying it to a decision run would replace the + DecisionTarget. It is not needed: DecisionTarget merges + ``TargetContext.trace_headers`` into its own request headers, so per-turn + traceparent propagation is not lost. + """ + import asyncio + + from simpleaudit.targets.base import TargetContext + + seen: list[httpx.Request] = [] + + def handler(request: httpx.Request) -> httpx.Response: + seen.append(request) + return httpx.Response(200, json=_recorded_reply()) + + target = decision_target_for(SNAPSHOT) + target._client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) + traceparent = "00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01" + asyncio.run( + target.send( + user=SCENARIO["description"], + context=TargetContext( + audit_run_id="audit_1", trace_headers={"traceparent": traceparent}, + extra={"decision": SCENARIO["decision"]}, + ), + ) + ) + assert seen[0].headers.get("traceparent") == traceparent diff --git a/infra/tests/test_decision_picker.py b/infra/tests/test_decision_picker.py new file mode 100644 index 00000000..9ba3266b --- /dev/null +++ b/infra/tests/test_decision_picker.py @@ -0,0 +1,91 @@ +"""The New Experiment model picker marks decision-capable models. + +Which models are decision models is settled on /connections/ (the probe writes +``capabilities.decision``); this is only the badge, so the picker renders from +``RegisteredModel.is_decision`` and nothing re-probes here. +""" + +from __future__ import annotations + +import pytest +from django.template.loader import render_to_string + +ROLES = [ + ("target", "Target", "The model under audit", []), + ("judge", "Judge", "Grades the answer", []), +] + + +class _Model: + """A registered model as the picker sees it.""" + + def __init__(self, pk, name, model_id, capabilities=None): + self.id = self.pk = pk + self.display_name = name + self.model_id = model_id + self.description = "" + self.has_key = True + self.capabilities = capabilities or {} + + @property + def is_decision(self) -> bool: + return bool((self.capabilities or {}).get("decision")) + + +class _Conn: + def __init__(self, pk, name, models): + self.pk = self.id = pk + self.name = name + self.description = "" + self.model_list = models + self.has_otlp = False + self.is_shared = False + self.share_label = "this workspace" + self.project = type("P", (), {"name": "Workspace"})() + + +def _render(models): + return render_to_string( + "partials/model_pickers.html", + { + "connections": [_Conn(1, "Local Ollama", models)], + "model_roles": ROLES, + "sel": {}, + "agent_models": [], + }, + ) + + +DECISION = _Model(1, "Clef Flash (local)", "clef-flash", {"decision": True}) +CHAT = _Model(2, "Qwen 3.5", "qwen3.5:2b") + + +def test_a_decision_model_is_badged(): + html = _render([DECISION]) + assert "Clef Flash (local)" in html + assert ">decision" in html + + +def test_a_chat_model_is_not_badged(): + html = _render([CHAT]) + assert "Qwen 3.5" in html + assert ">decision" not in html + + +def test_only_the_decision_model_is_badged_in_a_mixed_list(): + html = _render([DECISION, CHAT]) + assert html.count(">decision") == len(ROLES), "one badge per role, for one model" + + +@pytest.mark.parametrize("capabilities", [{}, {"decision": False}, {"vision": True}, None]) +def test_an_unprobed_or_negative_capability_gets_no_badge(capabilities): + html = _render([_Model(3, "Some model", "some-model", capabilities)]) + assert ">decision" not in html + + +def test_the_target_badge_says_a_run_is_one_turn(): + """The constraint a user needs before picking it as a target.""" + html = _render([DECISION]) + assert "single turn" in html + # The judge/auditor roles get the other explanation. + assert "cannot judge or drive an audit" in html diff --git a/infra/tests/test_decision_rep_view.py b/infra/tests/test_decision_rep_view.py new file mode 100644 index 00000000..0fce2045 --- /dev/null +++ b/infra/tests/test_decision_rep_view.py @@ -0,0 +1,84 @@ +"""The result page's Decision answer card. + +A decision target (SimpleAudit ``DecisionTarget``) answers once with a chosen +option, a probability for every option and a calibrated confidence. The auditor +stores that on the assistant reply as ``decision``; ``_rep_view`` lifts it onto +the rep so ``partials/result_rep_panel.html`` can render it. + +Scenarios without a decision block must be untouched, which is most of them. +""" +from django.template.loader import render_to_string +from django.test import TestCase + +from infra.ui import _rep_view + + +def _rep(answer=None, content="rights: patient rights"): + reply = {"role": "assistant", "content": content} + if answer is not None: + reply["decision"] = answer + return { + "severity": "pass", + "conversation": [{"role": "user", "content": "Classify this question."}, reply], + } + + +ANSWER = { + "choice": "rights", + "confidence": 0.82, + "probabilities": {"info": 0.18, "rights": 0.82}, +} + + +class DecisionRepViewTest(TestCase): + def test_decision_answer_is_lifted_onto_the_rep(self): + view = _rep_view(_rep(ANSWER), 1) + self.assertEqual(view["decision"]["choice"], "rights") + self.assertEqual(view["decision"]["confidence_percent"], 82.0) + self.assertEqual([o["option"] for o in view["decision"]["options"]], ["rights", "info"]) + + def test_no_decision_for_an_ordinary_conversation(self): + self.assertIsNone(_rep_view(_rep(), 1)["decision"]) + + def test_no_decision_for_an_empty_conversation(self): + self.assertIsNone(_rep_view({"severity": "pass", "conversation": []}, 1)["decision"]) + + def test_the_conversation_itself_is_unchanged(self): + """The reply still renders as a normal turn; the card is additional.""" + view = _rep_view(_rep(ANSWER), 1) + self.assertEqual(view["turns"], 1) + self.assertEqual(view["conversation"][1]["speaker"], "Target") + self.assertEqual(view["conversation"][1]["content"], "rights: patient rights") + + def test_last_answer_wins_when_several_replies_carry_one(self): + rep = { + "severity": "pass", + "conversation": [ + {"role": "assistant", "content": "a", "decision": {"choice": "a"}}, + {"role": "assistant", "content": "b", "decision": {"choice": "b"}}, + ], + } + self.assertEqual(_rep_view(rep, 1)["decision"]["choice"], "b") + + +class DecisionPanelRenderTest(TestCase): + """The partial renders the card, and stays silent without one.""" + + def _html(self, rep): + return render_to_string("partials/result_rep_panel.html", {"rep": _rep_view(rep, 1)}) + + def test_card_shows_choice_confidence_and_every_option(self): + html = self._html(_rep(ANSWER)) + self.assertIn("Decision answer", html) + self.assertIn("rights", html) + self.assertIn("confidence 82.0%", html) + self.assertIn("82.0%", html) + self.assertIn("18.0%", html) + + def test_no_card_without_a_decision_answer(self): + self.assertNotIn("Decision answer", self._html(_rep())) + + def test_card_without_probabilities_still_shows_the_choice(self): + html = self._html(_rep({"choice": "rights", "confidence": 0.5})) + self.assertIn("Decision answer", html) + self.assertIn("confidence 50.0%", html) diff --git a/infra/trace_target.py b/infra/trace_target.py index e91e95df..fef57455 100644 --- a/infra/trace_target.py +++ b/infra/trace_target.py @@ -295,3 +295,26 @@ class TraceContextModelAuditor(_SimpleAuditModelAuditor): def __init__(self, *args: Any, **kwargs: Any) -> None: super().__init__(*args, **kwargs) install_trace_context_target(self) + + +def decision_auditor_class(make_target: Any) -> type: + """A ModelAuditor subclass that sends to an explicit, non-model Target. + + The repetition runner builds its own auditors, so a decision run cannot + just call ``set_target`` on one instance — the class itself has to install + the target. ``make_target`` is called once per auditor, since a + ``DecisionTarget`` owns an HTTP client. + + No trace-context adapter here: that adapter wraps the OpenAI-compatible + chat client, which a decision target does not use. + """ + + class _DecisionModelAuditor(_SimpleAuditModelAuditor): + def __init__(self, *args: Any, **kwargs: Any) -> None: + from infra.engine import skipping_target_client + + with skipping_target_client(_SimpleAuditModelAuditor, skip=True): + super().__init__(*args, **kwargs) + self.set_target(make_target()) + + return _DecisionModelAuditor diff --git a/infra/ui.py b/infra/ui.py index 8a1976bc..ae792a4e 100644 --- a/infra/ui.py +++ b/infra/ui.py @@ -3006,6 +3006,13 @@ def _image_uris(rep: dict) -> list[str]: return out +def _decision_answer(message: object) -> dict | None: + """``model_registry.decision.decision_answer``, imported lazily for Django app loading.""" + from model_registry.decision import decision_answer + + return decision_answer(message) + + def _rep_view(rep: dict, index: int) -> dict: """One judged conversation (a repetition, or the whole single-rep result).""" def _token_count(value: object) -> int: @@ -3033,6 +3040,9 @@ def _token_count(value: object) -> int: "is_target": role == "assistant", "turn": turn, "content": (msg or {}).get("content", ""), + # A decision target answers in options, not prose: content is empty + # and the answer sits beside it (SimpleAudit 0.4.0). + "decision": _decision_answer(msg), }) tokens = [ { @@ -3047,8 +3057,12 @@ def _token_count(value: object) -> int: _token_count(t["input"]) + _token_count(t["output"]) for t in tokens ) grade = _judge_grade(rep.get("judgment")) + # The scenario's own decision answer: the last one in the conversation, so a + # follow-up turn's answer replaces the opening one. + decisions = [m["decision"] for m in conversation if m["decision"]] return { "index": index, + "decision": decisions[-1] if decisions else None, "severity": rep.get("severity", ""), "summary": rep.get("summary", ""), "grade": grade, diff --git a/model_registry/decision.py b/model_registry/decision.py new file mode 100644 index 00000000..2efe5796 --- /dev/null +++ b/model_registry/decision.py @@ -0,0 +1,139 @@ +"""Using a decision-capable registered model as an audit target. + +A decision model is not a chat model: it reads a state and answers a fixed +set of options with a choice, a probability per option and a calibrated +confidence — no prose, so no value to hallucinate. SimpleAudit 0.4.0 wraps +them in ``DecisionTarget``, with ``.ollama()`` for a local System One server +and ``.openrouter()`` for OpenRouter's Decisions API. + +Which registered models are decision models is already settled on +/connections/: the discover probe persists the flag on +``RegisteredModel.capabilities`` and ``RegisteredModel.is_decision`` reads it. +This module is the step after that — turning such a model, or the frozen +endpoint snapshot a run keeps of it, into the ``DecisionTarget`` the engine +sends to, and shaping the answer that comes back for run detail. +""" + +from __future__ import annotations + +from typing import Any + +#: The one provider whose decision endpoint is not ``{base_url}/v1/systemone``. +#: OpenRouter's Decisions API is a fixed, remote URL requiring a key, which is +#: why detection also treats it apart (model-id heuristic, not a probe). +OPENROUTER = "openrouter" + + +def is_decision_snapshot(snapshot: Any) -> bool: + """Whether a frozen endpoint snapshot is of a decision-capable model. + + The snapshot copies ``RegisteredModel.capabilities`` verbatim + (``audits.services._endpoint_snapshot``), so this is ``is_decision`` for a + run that has already been frozen — the live model may have been re-probed + or re-pointed since. + """ + if not isinstance(snapshot, dict): + return False + return bool((snapshot.get("capabilities") or {}).get("decision")) + + +def build_decision_target( + *, model_id: str, provider: str, base_url: str = "", api_key: str = "", + factory: Any = None, +) -> Any: + """A ``DecisionTarget`` for one model on one connection. + + The System One endpoint is a shared surface: every OpenAI-compatible + server that has one exposes it at ``{base_url}/v1/systemone`` (Ollama + >= 0.35, vLLM, …), which is why detection probes them all the same way + (``services.probe_systemone_decision``). Target setup needs no + per-provider endpoint logic either — only the auth handling. The + ``.ollama()`` factory is what builds that URL, and it also applies the + 64 KiB request-body cap. + + OpenRouter is the exception: a fixed remote URL that needs a key, which + Studio resolves from the connection's ``secret_reference`` (a raw key + never enters a snapshot). + + ``factory`` is for tests: anything with ``.ollama()`` / ``.openrouter()``. + The import stays inside the function so the registry keeps importing when + the installed engine predates 0.4.0. + """ + if factory is None: + from simpleaudit import DecisionTarget as factory # type: ignore[no-redef] + + if (provider or "").strip().lower() == OPENROUTER: + if not api_key: + raise ValueError( + "A decision model on OpenRouter needs an API key: name the environment " + "variable holding it in the connection's secret reference." + ) + return factory.openrouter(model_id, api_key=api_key) + + base = (base_url or "").strip().rstrip("/") + if not base: + raise ValueError("A decision model needs a base URL, e.g. http://localhost:11434.") + # The factory appends /v1/systemone, so a connection's own /v1 comes off + # first — the same normalisation the capability probe does. + extra = {"api_key": api_key} if api_key else {} + return factory.ollama(model_id, base_url=base.removesuffix("/v1"), **extra) + + +def decision_target_for_model(model: Any, *, api_key: str = "", factory: Any = None) -> Any: + """``build_decision_target`` for a live ``RegisteredModel``, gated on ``is_decision``.""" + if not getattr(model, "is_decision", False): + raise ValueError(f"{model} is not marked as a decision model.") + conn = model.connection + return build_decision_target( + model_id=model.model_id, provider=conn.provider, + base_url=conn.base_url or "", api_key=api_key, factory=factory, + ) + + +def decision_target_for_snapshot(snapshot: Any, *, api_key: str = "", factory: Any = None) -> Any | None: + """``build_decision_target`` for a frozen target snapshot, or ``None`` if it isn't one. + + Returning ``None`` rather than raising lets the engine call this on every + run and only divert the ones that are decision runs. + """ + if not is_decision_snapshot(snapshot): + return None + return build_decision_target( + model_id=snapshot.get("model_id") or "", provider=snapshot.get("provider") or "", + base_url=snapshot.get("base_url") or "", api_key=api_key, factory=factory, + ) + + +def decision_answer(message: Any) -> dict[str, Any] | None: + """The structured answer SimpleAudit stores beside an assistant reply. + + ``ModelAuditor`` copies ``TargetResponse.decision`` onto the reply as + ``decision``. Replies from a chat target have no such key, and this + returns ``None`` for them. Options come back sorted by probability so the + template can render them top-down without sorting in the view. + """ + if not isinstance(message, dict): + return None + answer = message.get("decision") + if not isinstance(answer, dict) or not answer.get("choice"): + return None + probs = answer.get("probabilities") + options = [] + if isinstance(probs, dict): + options = sorted( + ( + {"option": k, "probability": v, "percent": round(float(v) * 100, 1)} + for k, v in probs.items() + if isinstance(v, (int, float)) + ), + key=lambda o: -o["probability"], + ) + confidence = answer.get("confidence") + return { + "choice": answer.get("choice"), + "confidence": confidence, + "confidence_percent": ( + round(float(confidence) * 100, 1) if isinstance(confidence, (int, float)) else None + ), + "options": options, + } diff --git a/model_registry/tests/test_decision.py b/model_registry/tests/test_decision.py new file mode 100644 index 00000000..c48fc912 --- /dev/null +++ b/model_registry/tests/test_decision.py @@ -0,0 +1,216 @@ +"""Decision-capable models as audit targets: the gate, target construction, answer shaping. + +No network. ``build_decision_target`` takes a ``factory`` so the real +``DecisionTarget`` is never constructed against a live endpoint. +""" + +from __future__ import annotations + +import pytest + +from model_registry.decision import ( + build_decision_target, + decision_answer, + decision_target_for_model, + decision_target_for_snapshot, + is_decision_snapshot, +) + + +class _Conn: + def __init__(self, provider="", base_url="", secret_reference=""): + self.provider = provider + self.base_url = base_url + self.secret_reference = secret_reference + + def __str__(self): + return f"conn({self.provider})" + + +class _Model: + """Stands in for RegisteredModel, with the same ``is_decision`` semantics.""" + + def __init__(self, connection, model_id="clef-flash", capabilities=None): + self.connection = connection + self.model_id = model_id + self.capabilities = capabilities if capabilities is not None else {"decision": True} + + @property + def is_decision(self) -> bool: + return bool((self.capabilities or {}).get("decision")) + + def __str__(self): + return f"model({self.model_id})" + + +class _Factory: + """Stand-in for simpleaudit.DecisionTarget: records the call, builds nothing.""" + + def __init__(self): + self.calls = [] + + def ollama(self, model, base_url=None, **kw): + self.calls.append(("ollama", model, {"base_url": base_url, **kw})) + return ("ollama", model, base_url) + + def openrouter(self, model, api_key=None, **kw): + self.calls.append(("openrouter", model, {"api_key": api_key, **kw})) + return ("openrouter", model, api_key) + + +def _snap(**over): + snap = { + "model_id": "clef-flash", + "provider": "ollama", + "base_url": "http://localhost:11434/v1", + "capabilities": {"decision": True}, + "secret_reference": "", + } + snap.update(over) + return snap + + +# --- the capability gate -------------------------------------------------- + +def test_a_probed_decision_snapshot_is_recognised(): + assert is_decision_snapshot(_snap()) + + +@pytest.mark.parametrize("caps", [None, {}, {"decision": False}, {"vision": True}]) +def test_a_chat_snapshot_is_not(caps): + assert not is_decision_snapshot(_snap(capabilities=caps)) + + +@pytest.mark.parametrize("snapshot", [None, "text", 7, []]) +def test_malformed_snapshot_is_not_a_decision_snapshot(snapshot): + assert not is_decision_snapshot(snapshot) + + +def test_a_chat_snapshot_builds_no_target(): + """The engine calls this on every run, so a chat run must get None, not an error.""" + assert decision_target_for_snapshot(_snap(capabilities={}), factory=_Factory()) is None + + +# --- target construction -------------------------------------------------- + +def test_ollama_target_gets_the_server_root_not_the_v1_base(): + """DecisionTarget.ollama appends /v1/systemone itself, so /v1 must come off first.""" + f = _Factory() + decision_target_for_snapshot(_snap(), factory=f) + assert f.calls[0][0] == "ollama" + assert f.calls[0][1] == "clef-flash" + assert f.calls[0][2]["base_url"] == "http://localhost:11434" + + +def test_a_trailing_slash_is_also_stripped(): + f = _Factory() + decision_target_for_snapshot(_snap(base_url="http://localhost:11434/"), factory=f) + assert f.calls[0][2]["base_url"] == "http://localhost:11434" + + +def test_a_systemone_server_needs_a_base_url(): + with pytest.raises(ValueError, match="base URL"): + decision_target_for_snapshot(_snap(base_url=""), factory=_Factory()) + + +def test_openrouter_target_gets_the_resolved_key(): + f = _Factory() + decision_target_for_snapshot( + _snap(provider="openrouter", model_id="typesafe/jev-1.13", secret_reference="OPENROUTER_API_KEY"), + api_key="resolved-key", factory=f, + ) + assert f.calls[0][0] == "openrouter" + assert f.calls[0][1] == "typesafe/jev-1.13" + assert f.calls[0][2]["api_key"] == "resolved-key" + + +def test_openrouter_without_a_key_says_what_to_set(): + with pytest.raises(ValueError, match="secret reference"): + decision_target_for_snapshot(_snap(provider="openrouter"), api_key="", factory=_Factory()) + + +@pytest.mark.parametrize("provider", ["vllm", "openai", "simulachat", ""]) +def test_any_systemone_server_works_without_a_provider_allow_list(provider): + """The gate is capabilities.decision, not a list of known providers. + + Detection probes every non-OpenRouter server at the same endpoint + (f4e91d2), so a provider added upstream later needs no change here. + """ + f = _Factory() + build_decision_target( + model_id="clef-flash", provider=provider, base_url="http://vllm.internal:8000/v1", factory=f, + ) + assert f.calls[0][0] == "ollama" # the factory that builds {root}/v1/systemone + assert f.calls[0][2]["base_url"] == "http://vllm.internal:8000" + + +def test_a_key_is_forwarded_to_a_self_hosted_server_that_needs_one(): + f = _Factory() + build_decision_target( + model_id="clef-flash", provider="vllm", base_url="http://vllm.internal:8000/v1", + api_key="resolved-key", factory=f, + ) + assert f.calls[0][2]["api_key"] == "resolved-key" + + +def test_openrouter_matching_ignores_case_and_padding(): + f = _Factory() + build_decision_target( + model_id="typesafe/jev-1.13", provider=" OpenRouter ", api_key="k", factory=f, + ) + assert f.calls[0][0] == "openrouter" + + +# --- the live-model gate -------------------------------------------------- + +def test_a_registered_decision_model_builds_its_target(): + f = _Factory() + m = _Model(_Conn(provider="ollama", base_url="http://localhost:11434/v1")) + decision_target_for_model(m, factory=f) + assert f.calls[0][:2] == ("ollama", "clef-flash") + + +def test_a_model_not_marked_as_a_decision_model_is_refused(): + """is_decision is the gate: the /connections/ probe decides, not this module.""" + m = _Model(_Conn(provider="ollama", base_url="http://localhost:11434"), capabilities={}) + with pytest.raises(ValueError, match="not marked as a decision model"): + decision_target_for_model(m, factory=_Factory()) + + +# --- answer shaping ------------------------------------------------------- + +def test_answer_is_shaped_for_the_template(): + msg = { + "role": "assistant", + "content": "rights: patient rights", + "decision": { + "choice": "rights", + "confidence": 0.82, + "probabilities": {"info": 0.18, "rights": 0.82}, + }, + } + out = decision_answer(msg) + assert out["choice"] == "rights" + assert out["confidence_percent"] == 82.0 + # Highest probability first, so the chosen option reads at the top. + assert [o["option"] for o in out["options"]] == ["rights", "info"] + assert out["options"][0]["percent"] == 82.0 + + +def test_an_ordinary_reply_has_no_decision(): + assert decision_answer({"role": "assistant", "content": "Here is an answer."}) is None + + +@pytest.mark.parametrize("msg", [None, {}, "text", {"decision": {}}, {"decision": "x"}]) +def test_malformed_input_is_none_not_an_error(msg): + assert decision_answer(msg) is None + + +def test_missing_probabilities_still_gives_the_choice(): + out = decision_answer({"decision": {"choice": "a", "confidence": 0.5}}) + assert out["choice"] == "a" and out["options"] == [] + + +def test_a_missing_confidence_is_not_rendered_as_zero(): + out = decision_answer({"decision": {"choice": "a", "probabilities": {"a": 1.0}}}) + assert out["confidence_percent"] is None diff --git a/templates/partials/model_pickers.html b/templates/partials/model_pickers.html index 83bd52b9..d734b283 100644 --- a/templates/partials/model_pickers.html +++ b/templates/partials/model_pickers.html @@ -67,7 +67,14 @@