From 045074c56c7914cac6305791d51372718380456a Mon Sep 17 00:00:00 2001 From: SaladDay <1203511142@qq.com> Date: Thu, 8 Oct 2026 15:08:38 +0000 Subject: [PATCH] Qualify native public policies --- contracts/agents-api/harness-onboarding.md | 11 +- contracts/agents-api/zh/harness-onboarding.md | 13 +- .../tests/official_hosted_functions_native.py | 90 ++++++++--- .../official_hosted_structured_native.py | 77 +++++++-- .../core/tests/official_native_policies.py | 147 ++++++++++++++++++ .../core/tests/official_native_steering.py | 140 +++++++++++++++++ services/core/tests/qualify_public_native.py | 13 +- .../core/tests/qualify_public_native_test.py | 43 +++++ 8 files changed, 491 insertions(+), 43 deletions(-) create mode 100644 services/core/tests/official_native_policies.py create mode 100644 services/core/tests/official_native_steering.py diff --git a/contracts/agents-api/harness-onboarding.md b/contracts/agents-api/harness-onboarding.md index 8a22dc571..707db7a67 100644 --- a/contracts/agents-api/harness-onboarding.md +++ b/contracts/agents-api/harness-onboarding.md @@ -227,19 +227,24 @@ python services/core/tests/qualify_public_native.py \ | `none` | Creation retry, foreign history rejection, two native text Turns, history recall, SSE ordering and SDK/raw Item and Turn parity | `none` | | `pending-actions` | Query/reconnect pending calls, success/error results, cancellation, exact target rejection, retries and durable Items | Any declared placement with function tools | | `functions` | SDK handlers, success/error, native file and Artifact bytes, continuation, pending cancellation and tenant isolation | Workspace | +| `tool-search` | Deferred function execution, saved configuration freeze and the function suite | Workspace with declared deferred discovery | +| `policies` | Saved/inline disabled controls, unsupported enablement rejection, Codex native-parameter rejection and verbosity | Claude Code or Codex workspace | +| `steering` | Active message delivery, same-Turn attribution, public retries/conflicts, foreign rejection and continuation | MiniMax Code workspace | | `images` | Initial and active images, image results, retry/atomic rejection, native files/Artifacts, isolation and continuation | Workspace with declared image and function support | -| `structured` | Saved and inline schema, function-assisted native files, exact JSON/SSE, cancellation and text override | Workspace with declared structured output and function support | +| `structured` | Saved and inline schema, function-assisted native files, exact JSON/SSE, large integers, schema admission, cancellation and text override | Workspace with declared structured output and function support | | `composition` | Frozen source snapshots, initial bytes, ordered setup, packages, Skills, stdio MCP, continuation and cancellation | `openai_hosted` | For `composition`, settings must supply exactly `{"type":"openai_hosted"}` as `environment`; the suite creates its own Template, file and Skill sources. It checks initial binary bytes, ordered setup, npm and Python packages, uploaded, plugin and directory Skills, and three real stdio MCP tool identities, Items and results. After changing or deleting source resources, it verifies frozen preparation through warm continuation and, when selected, the existing Compose-verified agent-host restart. Cancellation must stop the MCP descendant's file effects. The suite cleans up all resources it creates. Private-owner and credential isolation remain `unverified`; positive canary evidence requires separate proof using operator-owned resources. The suite does not request `packages.system` or stdio MCP `env_vars`, which the current contracts reject. +The `tool-search` suite proves deferred callbacks through Core; the pinned public protocol has no required discovery event, so it does not claim a visible native ToolSearch call. `structured` distinguishes rejected lossy schema numbers from exact large-integer final text. `policies` records observed disabled-tool behavior, not network isolation or proof that a model honored verbosity; Codex low/high runs require native catalog support and an unsupported selection fails qualification. `steering` requires input while the original native Turn is active. Its public idempotency checks do not observe ACP duplicate receipts, and native acceptance does not guarantee model consumption. These limits remain in each suite’s evidence. + For `self_hosted`, choose a custom absolute `workspace_directory`. When the runner prints each new Session ID, connect a separate isolated machine or container using that Session's public installation command; the runner waits up to five minutes. Multiple Sessions must not share a workspace. The separate `official_environment_files_native.py` check accepts two already connected self-hosted Sessions and checks Files.list sorting, pagination and isolation through the Environment owner. -Warm continuation is the default and records cold recovery as `unverified`. To qualify cold continuation, additionally pass `--compose-directory` with the owned installation's absolute directory and `--compose-project` with its exact project name. The runner restarts only that project's `agent-host`, confirms its container start time changed, and then runs the unchanged history assertions. This does not qualify a Core restart or a sandbox checkpoint restore. The `pending-actions` suite reconnects the public client, not the agent-host process, and rejects those restart options. Select cold recovery only where the [declaration and coverage ledger](./index.md#known-gaps) support it; unsupported recovery remains a gap, never a successful skipped check. An API rejection fails the selected suite. +Warm continuation is the default and records cold recovery as `unverified`. To qualify cold continuation, additionally pass `--compose-directory` with the owned installation's absolute directory and `--compose-project` with its exact project name. The runner restarts only that project's `agent-host`, confirms its container start time changed, and then runs the unchanged history assertions. This does not qualify a Core restart or a sandbox checkpoint restore. The `pending-actions`, `policies` and `steering` suites do not qualify process restart and reject those options. Select cold recovery only where the [declaration and coverage ledger](./index.md#known-gaps) support it; unsupported recovery remains a gap, never a successful skipped check. An API rejection fails the selected suite. -Run `python -m unittest discover -s services/core/tests -p qualify_public_native_test.py` with the pinned SDK to check credential handling and the owned restart boundary without a model. Existing deterministic Core integration tests remain the authority for schema validation, atomic admission, durable receipts and rejection semantics. Real-model results qualify only the selected suite, protocol, Harness and placement. Provider lifecycle, native identity, credential isolation, deferred discovery and unselected suites need separate evidence; view-only results do not qualify the public path. +Run `python -m unittest discover -s services/core/tests -p qualify_public_native_test.py` with the pinned SDK to check credential handling and the owned restart boundary without a model. Existing deterministic Core integration tests remain the authority for schema validation, atomic admission, durable receipts and rejection semantics. Real-model results qualify only the selected suite, protocol, Harness and placement. Provider lifecycle, native identity, credential isolation and unselected suites need separate evidence; view-only results do not qualify the public path. ## Native version pins diff --git a/contracts/agents-api/zh/harness-onboarding.md b/contracts/agents-api/zh/harness-onboarding.md index e4e271fa3..7e33fba6e 100644 --- a/contracts/agents-api/zh/harness-onboarding.md +++ b/contracts/agents-api/zh/harness-onboarding.md @@ -1,7 +1,7 @@ --- title: "添加 Harness" source: contracts/agents-api/harness-onboarding.md -source_hash: 32b85617efe41a23bd25768a42fd7a57019c75ae723d07d8cc5106124709f402 +source_hash: 1fc4bf541841e7c955cbea4ff4d2d903132a744cfa37a972a918ff1e692e1cba --- **Harness** 是一种运行模型和工具循环的原生代理引擎(Codex、Claude Code、MiniMax Code)。**Harness 适配器**将 Runtime 的 Executor 和 Turn 契约转换到该引擎的 SDK 或协议。本文档定义 Runtime–Harness 协议:适配器接口及其生命周期义务、注册、支持声明和验收。 @@ -229,19 +229,24 @@ python services/core/tests/qualify_public_native.py \ | `none` | 创建重试、外部历史拒绝、两个原生文本 Turn、历史回忆、SSE 顺序,以及 SDK/原始 Item 和 Turn 一致性 | `none` | | `pending-actions` | 查询和重连待处理调用、成功/错误结果、取消、精确目标拒绝、重试和持久化 Item | 声明支持函数工具的任意放置方式 | | `functions` | SDK handler、成功/错误、原生文件和 Artifact 字节、继续执行、待处理调用取消和租户隔离 | 工作区 | +| `tool-search` | 延迟函数执行、已保存配置冻结及函数套件 | 声明支持延迟发现的工作区 | +| `policies` | 保存/内联禁用控制、不支持的启用拒绝、Codex 原生参数拒绝及 verbosity | Claude Code 或 Codex 工作区 | +| `steering` | 活跃消息投递、同一 Turn 归属、公共重试/冲突、跨租户拒绝及继续执行 | MiniMax Code 工作区 | | `images` | 初始和活动图像、图像结果、重试/原子拒绝、原生文件/Artifact、隔离和继续执行 | 声明支持图像和函数的工作区 | -| `structured` | 保存和内联 schema、函数辅助原生文件、精确 JSON/SSE、取消和文本覆盖 | 声明支持结构化输出和函数的工作区 | +| `structured` | 保存和内联 schema、函数辅助原生文件、精确 JSON/SSE、大整数、schema 准入、取消和文本覆盖 | 声明支持结构化输出和函数的工作区 | | `composition` | 冻结的源快照、初始字节、有序 setup、包、Skills、stdio MCP、继续执行和取消 | `openai_hosted` | 对于 `composition`,设置中的 `environment` 必须恰好为 `{"type":"openai_hosted"}`;套件创建自己的 Template、文件和 Skill 来源。它检查初始二进制字节、有序 setup、npm 和 Python 包、上传的 Skill、插件 Skill 和目录 Skill,以及三个真实 stdio MCP 工具的身份、Item 和结果。更改或删除源资源后,它通过热继续执行验证冻结的准备结果;选用重启时,还通过现有的 Compose 验证 agent-host 重启流程进行检查。取消必须使 MCP 后代进程停止产生文件副作用。套件清理其创建的全部资源。 私有所有者与凭据隔离仍为 `unverified`;肯定性的 canary 证据需要使用操作员拥有的资源独立验证。套件不请求当前契约拒绝的 `packages.system` 或 stdio MCP `env_vars`。 +`tool-search` 套件证明延迟回调通过 Core 执行;锁定的公共协议没有必需的发现事件,因此它不声称观察到了原生 ToolSearch 调用。`structured` 区分有损 schema 数字的拒绝与精确保留的大整数最终文本。`policies` 记录观察到的工具禁用行为,不代表网络隔离,也不能证明模型遵循了 verbosity;Codex 的 low/high 执行需要原生 catalog 支持,不支持的选择会使验收失败。`steering` 要求输入提交时原始原生 Turn 仍活跃。其公共幂等性检查不观察 ACP 重复回执,原生接收也不保证模型采用输入。各套件的证据保留这些限制。 + 对于 `self_hosted`,选择自定义绝对 `workspace_directory`。运行器打印每个新 Session ID 后,使用该 Session 的公共安装命令连接独立的隔离机器或容器;运行器最多等待五分钟。多个 Session 不得共享工作区。独立的 `official_environment_files_native.py` 检查接受两个已连接的 self-hosted Session,通过 Environment owner 验证 Files.list 排序、分页和隔离。 -默认验证热继续执行,并将冷恢复记录为 `unverified`。验证冷继续执行时,额外传入指向所拥有安装的绝对目录的 `--compose-directory`,以及指定其精确项目名称的 `--compose-project`。运行器仅重启该项目的 `agent-host`,确认容器启动时间已改变,然后执行相同的历史断言。这不能证明 Core 重启或 sandbox 检查点恢复。`pending-actions` 套件重连的是公共客户端,而非 agent-host 进程,因此拒绝这些重启选项。仅在[声明和覆盖台账](./index.md#known-gaps) 支持时选择冷恢复;不支持的恢复仍是缺口,不能把跳过的检查记为成功。API 拒绝会使所选套件失败。 +默认验证热继续执行,并将冷恢复记录为 `unverified`。验证冷继续执行时,额外传入指向所拥有安装的绝对目录的 `--compose-directory`,以及指定其精确项目名称的 `--compose-project`。运行器仅重启该项目的 `agent-host`,确认容器启动时间已改变,然后执行相同的历史断言。这不能证明 Core 重启或 sandbox 检查点恢复。`pending-actions`、`policies` 和 `steering` 套件不验证进程重启,因此拒绝这些选项。仅在[声明和覆盖台账](./index.md#known-gaps) 支持时选择冷恢复;不支持的恢复仍是缺口,不能把跳过的检查记为成功。API 拒绝会使所选套件失败。 -使用锁定的 SDK 运行 `python -m unittest discover -s services/core/tests -p qualify_public_native_test.py`,可在不调用模型的情况下检查凭据处理和所拥有的重启边界。现有确定性 Core 集成测试仍负责 schema 验证、原子准入、持久化回执和拒绝语义。真实模型结果仅证明所选套件、协议、Harness 和放置方式。Provider 生命周期、原生身份、凭据隔离、延迟发现和未选择的套件需要独立证据;仅通过 view 测试不能证明公共调用路径。 +使用锁定的 SDK 运行 `python -m unittest discover -s services/core/tests -p qualify_public_native_test.py`,可在不调用模型的情况下检查凭据处理和所拥有的重启边界。现有确定性 Core 集成测试仍负责 schema 验证、原子准入、持久化回执和拒绝语义。真实模型结果仅证明所选套件、协议、Harness 和放置方式。Provider 生命周期、原生身份、凭据隔离和未选择的套件需要独立证据;仅通过 view 测试不能证明公共调用路径。 ## 原生版本固定 {#native-version-pins} diff --git a/services/core/tests/official_hosted_functions_native.py b/services/core/tests/official_hosted_functions_native.py index c4df8aa7c..5dbc8b1b8 100644 --- a/services/core/tests/official_hosted_functions_native.py +++ b/services/core/tests/official_hosted_functions_native.py @@ -9,7 +9,7 @@ from session_cleanup import delete_session -def verify_hosted_functions(client, foreign, http, agent_options, session_options, ready, restart, record): +def verify_hosted_functions(client, foreign, http, agent_options, session_options, ready, restart, record, *, deferred=False): """Run against the selected workspace; restart, when supplied, restarts its agent host.""" pin = json.loads((Path(__file__).resolve().parents[3] / "contracts/agents-api/upstream.json").read_text()) distribution = importlib.metadata.distribution("openai") @@ -22,10 +22,20 @@ def verify_hosted_functions(client, foreign, http, agent_options, session_option marker = "function-memory-" + uuid.uuid4().hex private_error = "private-handler-" + uuid.uuid4().hex calls, observed, checks = [], [], [] - session = sessions.create(agent={**agent_options, "instructions": "Follow the requested tool calls exactly. Never repeat a failed call.", "tools": [{ + tool = { "type": "function", "name": "lookup", "description": "Retrieve the requested test value.", "parameters": {"type": "object", "properties": {"key": {"type": "string"}}, "required": ["key"], "additionalProperties": False}, - }]}, **session_options) + } + if deferred: + tool["defer_loading"] = True + tools = [{"type": "tool_search"}, tool] if deferred else [tool] + configuration = {**agent_options, "instructions": "Follow the requested tool calls exactly. Never repeat a failed call.", "tools": tools} + owned, saved = [], None + proof = {"checks": checks, "calls": calls, "events": observed, "deferred": deferred, + "model": agent_options["model"], "harness": agent_options["x_agents_core"]["harness"]} + if deferred: + proof["evidence_limit"] = "Deferred callback execution; the public stream does not independently expose native ToolSearch discovery." + discovery = "Search your tools for the function that retrieves the requested test value, then " if deferred else "" def until(predicate, timeout=120): deadline = time.monotonic() + timeout @@ -44,9 +54,12 @@ def items(): return raw def invoke(text, key, handler): - with sessions.stream(session.id, input=text, tool_handlers={"lookup": handler}, idempotency_key=key, timeout=240) as stream: - events = [event.to_dict() for event in stream] + events = [] observed.append(events) + if saved is not None: + proof["saved"]["events"].append(events) + with sessions.stream(session.id, input=text, tool_handlers={"lookup": handler}, idempotency_key=key, timeout=240) as stream: + events.extend(event.to_dict() for event in stream) terminals = [e for e in events if e["type"] in ("agent.session.turn.completed", "agent.session.turn.failed", "agent.session.turn.cancelled")] assert len(terminals) == 1 and terminals[0]["type"] == "agent.session.turn.completed" assert events[-1]["type"] == "agent.session.idle" @@ -58,7 +71,22 @@ def invoke(text, key, handler): assert sessions.retrieve(session.id).required_actions == [] return terminals[0]["turn"]["id"], result[0]["item"] + def successful_file(path, key): + turn, result = invoke(discovery + f"Call lookup once with key success. Remember its returned string in conversation. Then use the native shell to create {workspace}/outputs/{path} containing exactly that string, with no newline. Do not call any other function.", key, success) + assert result["output"] == marker and "error" in result and result["error"] is None + artifacts = list(sessions.artifacts.list(session.id, limit=100)) + output = [a for a in artifacts if a.turn_id == turn and a.path == "/workspace/outputs/" + path] + assert len(output) == 1 + with sessions.artifacts.with_streaming_response.content(output[0].id, session_id=session.id) as response: + assert response.read() == marker.encode() + listing = client.beta.agents.environments.files.list(session.environment.id, path="/workspace/outputs") + assert any(f.path == "/workspace/outputs/" + path for f in listing) + assert any(i["type"] == "function_call_output" and i.get("output") == marker for i in items()) + try: + session = sessions.create(agent=configuration, **session_options) + owned.append(session.id) + proof["session"] = session.id ready(session) def success(arguments): @@ -66,16 +94,7 @@ def success(arguments): calls.append(arguments) return marker - turn, result = invoke(f"Call lookup once with key success. Remember its returned string in conversation. Then use the native shell to create {workspace}/outputs/function.txt containing exactly that string, with no newline. Do not call any other function.", "function-success", success) - assert result["output"] == marker and "error" in result and result["error"] is None - artifacts = list(sessions.artifacts.list(session.id, limit=100)) - output = [a for a in artifacts if a.turn_id == turn and a.path == "/workspace/outputs/function.txt"] - assert len(output) == 1 - with sessions.artifacts.with_streaming_response.content(output[0].id, session_id=session.id) as response: - assert response.read() == marker.encode() - listing = client.beta.agents.environments.files.list(session.environment.id, path="/workspace/outputs") - assert any(f.path == "/workspace/outputs/function.txt" for f in listing) - assert any(i["type"] == "function_call_output" and i.get("output") == marker for i in items()) + successful_file("function.txt", "function-success") checks.append("same_turn_function_native_file_and_public_artifact") if restart is not None: @@ -86,14 +105,14 @@ def failure(arguments): calls.append(arguments) raise RuntimeError(private_error) - _, result = invoke("Call lookup exactly once with key failure. If it fails, do not retry. Then reply with the exact string returned by lookup in our first turn, using conversation history and no file tools.", "function-failure", failure) + _, result = invoke(discovery + "Call lookup exactly once with key failure. If it fails, do not retry. Then reply with the exact string returned by lookup in our first turn, using conversation history and no file tools.", "function-failure", failure) assert result["error"] == "Tool handler failed." and "output" in result and result["output"] is None messages = [i for i in items() if i["type"] == "message" and i["role"] == "assistant"] assert marker in "\n".join(p.get("text", "") for p in messages[-1]["content"]) assert len(calls) == 2 checks.append(("cold" if restart is not None else "warm") + "_history_continuation_and_public_handler_error") - sessions.events.create(session.id, events=[{"type": "agent.session.input.message", "input": [{"role": "user", "content": [{"type": "input_text", "text": "Call lookup once with key pending and wait for the result."}]}]}], idempotency_key="function-pending") + sessions.events.create(session.id, events=[{"type": "agent.session.input.message", "input": [{"role": "user", "content": [{"type": "input_text", "text": discovery + "Call lookup once with key pending and wait for the result."}]}]}], idempotency_key="function-pending") until(lambda: sessions.retrieve(session.id).required_actions) pending = [i for i in items() if i["type"] == "function_call"][-1] foreign_headers = {**headers, "Authorization": "Bearer " + foreign.api_key} @@ -107,9 +126,38 @@ def failure(arguments): assert sessions.retrieve(session.id).required_actions == [] assert not any(i["type"] == "function_call_output" and i["call_id"] == pending["call_id"] for i in items()) checks.append("pending_function_tenant_isolation_and_cancel_retry") - proof = {"checks": checks, "session": session.id, "calls": calls, "events": observed, "items": items()} - assert private_error not in json.dumps(proof) - record(proof) + proof.update(events=list(observed), items=items()) + if deferred: + record(proof) + saved = client.beta.agents.create(**{k: v for k, v in configuration.items() if k != "x_agents_core"}, + extra_body={"x_agents_core": configuration["x_agents_core"]}) + proof["saved"] = {"agent": saved.id, "original_tools": tools, "events": []} + stored = client.beta.agents.retrieve(saved.id) + assert [entry.to_dict() for entry in stored.tools] == tools + session = sessions.create(agent_id=saved.id, **session_options) + owned.append(session.id) + proof["saved"]["session"] = session.id + ready(session) + changed_tool = {**tool, "description": "Changed after Session creation.", + "parameters": {"type": "object", "properties": {"replacement": {"type": "string"}}, "required": ["replacement"], "additionalProperties": False}} + client.beta.agents.update(saved.id, tools=[{"type": "tool_search"}, changed_tool]) + assert [entry.to_dict() for entry in client.beta.agents.retrieve(saved.id).tools] == [{"type": "tool_search"}, changed_tool] + # The pinned Session tool union omits tool_search; the frozen + # deferred function and its original schema must still be present. + frozen = [entry.to_dict() for entry in sessions.retrieve(session.id).agent.tools] + assert frozen == [tool] + marker = "saved-function-memory-" + uuid.uuid4().hex + successful_file("saved-function.txt", "saved-function-success") + assert len(calls) == 3 and calls[-1] == {"key": "success"} + checks.append("saved_deferred_function_schema_frozen_after_agent_update") + proof["saved"].update(updated_tools=[{"type": "tool_search"}, changed_tool], session_tools=frozen, items=items()) return checks finally: - delete_session(sessions, session.id) + try: + assert private_error not in json.dumps(proof) + record(proof) + finally: + for sid in reversed(owned): + delete_session(sessions, sid) + if saved is not None: + client.beta.agents.delete(saved.id) diff --git a/services/core/tests/official_hosted_structured_native.py b/services/core/tests/official_hosted_structured_native.py index 9ec7bcf1f..8245d51d6 100644 --- a/services/core/tests/official_hosted_structured_native.py +++ b/services/core/tests/official_hosted_structured_native.py @@ -1,10 +1,13 @@ """Structured final answers after real native workspace/tool execution.""" import importlib.metadata +from decimal import Decimal import json from pathlib import Path import time import uuid +from openai import BadRequestError + from session_cleanup import delete_session @@ -18,7 +21,7 @@ def verify_hosted_structured(client, foreign, http, agent_options, session_optio root = str(client.base_url).rstrip("/") + "/agents" headers = {"Authorization": "Bearer " + client.api_key, "OpenAI-Beta": "agents=v1"} other_headers = {**headers, "Authorization": "Bearer " + foreign.api_key} - proof = {"engine": agent_options["x_agents_core"]["harness"], "checks": [], "runs": [], "calls": []} + proof = {"engine": agent_options["x_agents_core"]["harness"], "model": agent_options["model"], "checks": [], "runs": [], "calls": []} owned = [] saved = None schema = {"type": "object", "properties": {name: {"type": "string"} for name in ("memory", "path", "marker")}, @@ -83,8 +86,9 @@ def final(sid, eid, turn, expected, events=None): answers = [i for i in stored if i["type"] == "message" and i.get("role") == "assistant" and i.get("phase") == "final_answer" and i["turn_id"] == turn.id] assert len(answers) == 1, answers answer = answers[0] + proof.setdefault("answers", []).append(answer) raw = answer["content"][0]["text"] - assert json.loads(raw) == expected, raw + assert json.loads(raw, parse_float=Decimal) == expected, raw if events: added = next(i for i, e in enumerate(events) if e["type"] == "agent.session.turn.item.added" and e["item"]["id"] == answer["id"]) done = next(i for i, e in enumerate(events) if e["type"] == "agent.session.turn.item.done" and e["item"]["id"] == answer["id"]) @@ -95,15 +99,16 @@ def final(sid, eid, turn, expected, events=None): assert events[added]["item"]["status"] == "in_progress" and events[added]["item"]["content"] == [] deltas = [e["delta"] for e in events if e["type"] == "agent.session.turn.output_text.delta" and e["item_id"] == answer["id"]] assert deltas and "".join(deltas) == raw, deltas - assert expected["path"].startswith(workspace + "/") - public_path = "/workspace" + expected["path"][len(workspace):] - artifacts = [a for a in sessions.artifacts.list(sid, limit=100) if a.path == public_path and a.turn_id == turn.id] - assert len(artifacts) == 1 - with sessions.artifacts.with_streaming_response.content(artifacts[0].id, session_id=sid) as response: - assert response.read() == expected["memory"].encode() - assert any(f.path == public_path for f in client.beta.agents.environments.files.list(eid, path="/workspace/outputs")) - assert http.get(root + "/sessions/" + sid + "/artifacts/" + artifacts[0].id + "/content", headers=other_headers).status_code == 404 - proof.setdefault("answers", []).append(answer) + if "path" in expected: + assert expected["path"].startswith(workspace + "/") + public_path = "/workspace" + expected["path"][len(workspace):] + artifacts = [a for a in sessions.artifacts.list(sid, limit=100) if a.path == public_path and a.turn_id == turn.id] + assert len(artifacts) == 1 + with sessions.artifacts.with_streaming_response.content(artifacts[0].id, session_id=sid) as response: + assert response.read() == expected["memory"].encode() + assert any(f.path == public_path for f in client.beta.agents.environments.files.list(eid, path="/workspace/outputs")) + assert http.get(root + "/sessions/" + sid + "/artifacts/" + artifacts[0].id + "/content", headers=other_headers).status_code == 404 + return raw def run(sid, text, handler=None): events = [] @@ -212,6 +217,56 @@ def cancel(action): final(inline.id, inline.environment.id, terminal(inline.id, 1), {"memory": inline_memory, "path": path, "marker": "inline"}, events) assert inline.agent.text.format.to_dict() == output_format check("inline_schema_prepared_execution_native_file_and_final_json") + numeric_schema = {"type": "object", "properties": {"n": {"type": "integer"}}, + "required": ["n"], "additionalProperties": False} + exact_schema = {**numeric_schema, "properties": {"n": {"type": "integer", "const": 9007199254740992}}} + for source, selected_schema, number in (("saved", exact_schema, 9007199254740992), + ("inline", exact_schema, 9007199254740992), + ("inline", numeric_schema, 9007199254740993)): + selected_format = {"type": "json_schema", "schema": selected_schema} + if source == "saved": + client.beta.agents.update(saved.id, tools=[], text={"format": selected_format}) + assert client.beta.agents.retrieve(saved.id).text.format.to_dict() == selected_format + numeric = sessions.create(agent_id=saved.id, **session_options) + else: + numeric = sessions.create(agent={**agent_options, "text": {"format": selected_format}}, **session_options) + owned.append(numeric.id) + ready(numeric) + assert sessions.retrieve(numeric.id).agent.text.format.to_dict() == selected_format + proof.setdefault("numeric", []).append({"source": source, "session": numeric.id, "format": selected_format, + "expected_integer": number, "model_dependent_output": True}) + events = run(numeric.id, f"Return the JSON object with key n and the exact integer {number}. Preserve every digit and use the required output format. Do not use workspace tools or public functions.") + raw = final(numeric.id, numeric.environment.id, terminal(numeric.id, 1), {"n": number}, events) + assert sessions.retrieve(numeric.id).agent.text.format.to_dict() == selected_format + proof["numeric"][-1]["raw_final_text"] = raw + check(source + ("_exact_binary64_schema_constant" if selected_schema is exact_schema else "_large_integer_final_text_sse_and_storage")) + + # This is the selected Claude adapter's schema admission policy, not + # a restriction on other Harnesses' numeric output or saved resources. + if proof["engine"] == "claude_sdk": + unsafe_format = {"type": "json_schema", "schema": {**numeric_schema, + "properties": {"n": {"type": "integer", "const": 9007199254740993}}}} + client.beta.agents.update(saved.id, tools=[], text={"format": unsafe_format}) + assert client.beta.agents.retrieve(saved.id).text.format.to_dict() == unsafe_format + expected_error = {"type": "invalid_request_error", "code": "unsupported_or_invalid_configuration", + "param": "agent.text.format", "message": "This runtime requires an object schema with lossless JSON numbers."} + for source, configuration in (("inline", {"agent": {**agent_options, "text": {"format": unsafe_format}}}), + ("saved", {"agent_id": saved.id})): + request = {**configuration, **session_options} + wire = {**{k: v for k, v in request.items() if k != "extra_body"}, **request.get("extra_body", {})} + response = http.post(root + "/sessions", headers=headers, json=wire) + if response.status_code == 201: + owned.append(response.json()["id"]) + assert response.status_code == 400 and response.json() == {"error": expected_error}, response.text + try: + unexpected = sessions.create(**request) + except BadRequestError as error: + assert error.status_code == 400 and error.body == expected_error, error.body + else: + owned.append(unexpected.id) + raise AssertionError("Lossy numeric schema was admitted") + proof.setdefault("numeric_rejections", []).append({"source": source, "format": unsafe_format, "error": response.json()}) + check(source + "_lossy_numeric_schema_rejected_before_execution") proof.update(passed=True, items=items(sid), turns=[t.to_dict() for t in sessions.turns.list(sid, order="asc", limit=100).data]) return proof["checks"] finally: diff --git a/services/core/tests/official_native_policies.py b/services/core/tests/official_native_policies.py new file mode 100644 index 000000000..21a11bf88 --- /dev/null +++ b/services/core/tests/official_native_policies.py @@ -0,0 +1,147 @@ +"""Public disabled-policy and Codex configuration checks through a real native Turn.""" + +from contextlib import ExitStack +import uuid + +from session_cleanup import delete_session + + +def verify_native_policies(client, foreign, http, agent_options, session_options, ready, record): + agents, sessions = client.beta.agents, client.beta.agents.sessions + harness = agent_options["x_agents_core"]["harness"] + assert harness in {"claude_sdk", "codex"}, "Policies qualify Claude Code or Codex" + root = str(client.base_url).rstrip("/") + "/agents" + headers = {"Authorization": "Bearer " + client.api_key, "OpenAI-Beta": "agents=v1"} + foreign_headers = {**headers, "Authorization": "Bearer " + foreign.api_key} + disabled = [{"type": "web_search", "mode": "disabled"}, {"type": "programmatic_tool_calling", "enabled": False}] + resolved = [{**disabled[0], "context_size": "medium", "allowed_domains": None, "location": None}, disabled[1]] + instructions = "Follow the requested answer format. Do not substitute another tool for an unavailable tool." + agent = {**agent_options, "instructions": instructions, "tools": disabled} + proof = {"harness": harness, "model": agent_options["model"], "checks": [], "runs": [], "rejections": [], "disabled_controls": resolved, + "unverified": [ + "Public configuration and absent tool events do not prove native policy enforcement or tool inventory; model self-reports are not enforcement evidence.", + "Disabled web search does not disable ordinary shell HTTP or establish network isolation.", + "Native catalog support and launch overrides are not collected here; completed verbosity Turns do not prove the model honored a verbosity level.", + "Native managed-policy injection and provider lifecycle require separate qualification."]} + + def reject(label, spec, message=None): + before = {session.id for session in sessions.list()} + body = {"environment": session_options["environment"], **session_options.get("extra_body", {}), **spec} + response = http.post(root + "/sessions", headers=headers, json=body) + # Track an incorrectly admitted Session so failure still cleans it up. + if response.status_code == 201: + cleanup.callback(delete_session, sessions, response.json()["id"]) + error = response.json().get("error", {}) + proof["rejections"].append({"case": label, "status": response.status_code, "error": error}) + record(proof) + assert response.status_code == 400 and error.get("type") == "invalid_request_error", label + if message is not None: + assert error.get("message") == message and error.get("code") == "unsupported_or_invalid_configuration" and error.get("param") is None, label + assert {session.id for session in sessions.list()} == before, "Rejected configuration created a Session" + proof["checks"].append(label) + + def run(session, label, prompt, marker, verbosity="medium"): + sid = session.id + run_proof = {"case": label, "session": sid, "prompt": prompt, "verbosity": verbosity, "events": [], "status": "running", "step": "readiness", "agent": session.agent.to_dict()} + if verbosity in {"low", "high"}: + run_proof["prerequisite"] = "The native model catalog must declare verbosity support; catalog contents are not independently observed here. A failed run does not identify the failure cause." + proof["runs"].append(run_proof) + try: + ready(session) + run_proof["step"] = "native_turn" + assert session.agent.instructions == instructions and session.agent.model == agent_options["model"] + assert session.agent.text.verbosity == verbosity + assert [tool.to_dict() for tool in session.agent.tools] == resolved + with sessions.stream(sid, input=prompt, timeout=240) as stream: + for event in stream: + run_proof["events"].append(event.to_dict()) + assert event.type not in {"agent.session.requires_action", "agent.session.failed", "agent.session.turn.failed", "agent.session.turn.cancelled"}, label + " requires a completed native Turn; unsupported verbosity is not a pass" + events = run_proof["events"] + types = [event["type"] for event in events] + assert types.count("agent.session.turn.created") == types.count("agent.session.turn.completed") == 1 + assert types[-1] == "agent.session.idle" and types.index("agent.session.turn.created") < types.index("agent.session.turn.completed") < len(types) - 1 + assert len({event["event_id"] for event in events}) == len(events) + run_proof["step"] = "public_persistence" + stored = {} + for resource in ("items", "turns"): + response = http.get(root + "/sessions/" + sid + "/" + resource, headers=headers, params={"order": "asc", "limit": 100}) + assert response.status_code == 200 and not response.json()["has_more"] + assert response.json() == getattr(sessions, resource).list(sid, order="asc", limit=100).to_dict() + stored[resource] = response.json()["data"] + assert len(stored["turns"]) == 1 and stored["turns"][0]["status"] == "completed" + observed_items = stored["items"] + [event["item"] for event in events if "item" in event] + assert not any(item["type"] in {"web_search_call", "function_call", "function_call_output", "mcp_call"} for item in observed_items) + answers = [item for item in stored["items"] if item["type"] == "message" and item.get("role") == "assistant" and item["turn_id"] == stored["turns"][0]["id"]] + assert answers and answers[-1]["status"] == "completed" + answer = answers[-1] + text = "".join(part.get("text", "") for part in answer["content"]) + assert text.strip() == marker, "Model did not produce the requested fixed answer" + done = [event for event in events if event["type"] == "agent.session.turn.output_text.done" and event["item_id"] == answer["id"]] + assert len(done) == 1 and done[0]["text"] == text + assert "".join(event["delta"] for event in events if event["type"] == "agent.session.turn.output_text.delta" and event["item_id"] == answer["id"]) == text + assert [item for item in stored["items"] if item["type"] == "message" and item.get("role") == "user"][0]["content"] == [{"type": "input_text", "text": prompt}] + current = sessions.retrieve(sid) + assert current.required_actions == [] and current.agent == session.agent + for suffix in ("", "/items", "/turns"): + assert http.get(root + "/sessions/" + sid + suffix, headers=foreign_headers).status_code == 404 + run_proof.update(status="passed", **stored) + proof["checks"].append(label) + finally: + proof["current_case"] = {"case": label, "step": run_proof["step"], "verbosity": verbosity} + if run_proof["status"] != "passed": + run_proof["status"] = "failed" + record(proof) + + with ExitStack() as cleanup: + try: + for label, tool, message in ( + ("web_search_live_rejected", {"type": "web_search", "mode": "live"}, "Only disabled web_search is qualified for execution."), + ("web_search_cached_rejected", {"type": "web_search", "mode": "cached"}, "Only disabled web_search is qualified for execution."), + ("web_search_default_rejected", {"type": "web_search"}, "Only disabled web_search is qualified for execution."), + ("programmatic_enabled_rejected", {"type": "programmatic_tool_calling", "enabled": True}, "Programmatic tool calling is not qualified for execution."), + ): + reject(label, {"agent": {**agent, "tools": [tool]}}, message) + saved = agents.create(**{key: value for key, value in agent.items() if key != "x_agents_core"}, extra_body={"x_agents_core": agent["x_agents_core"]}) + cleanup.callback(agents.delete, saved.id) + assert [tool.to_dict() for tool in agents.retrieve(saved.id).tools] == resolved + frozen = sessions.create(agent_id=saved.id, **session_options) + cleanup.callback(delete_session, sessions, frozen.id) + # Saved configuration admits enabled controls; only Session admission rejects them. + for tool, message in (({"type": "web_search", "mode": "live"}, "Only disabled web_search is qualified for execution."), + ({"type": "programmatic_tool_calling", "enabled": True}, "Programmatic tool calling is not qualified for execution.")): + updated = agents.update(saved.id, tools=[tool]) + expected = {**tool, "context_size": "medium", "allowed_domains": None, "location": None} if tool["type"] == "web_search" else tool + assert [value.to_dict() for value in updated.tools] == [expected] + assert agents.retrieve(saved.id) == updated + assert sessions.retrieve(frozen.id).agent == frozen.agent + reject("saved_" + tool["type"] + "_enabled_rejected", {"agent_id": saved.id}, message) + inline = sessions.create(agent=agent, **session_options) + cleanup.callback(delete_session, sessions, inline.id) + for label, session in (("saved_disabled_controls_frozen", frozen), ("inline_disabled_controls", inline)): + marker = "POLICY_" + uuid.uuid4().hex + prompt = "Attempt native web search and programmatic tool calling. If either tool is unavailable, do not substitute tools. Regardless of availability, answer with exactly " + marker + "." + run(session, label, prompt, marker) + if harness == "codex": + for key, value in (("features.code_mode", True), ("features.hooks", True), ("features.plugins", True), + ("web_search", "live"), ("model_verbosity", "high"), ("tools.experimental_request_user_input.enabled", True), + ("model_reasoning_effort", True), ("model_reasoning_effort", "unknown")): + extension = {**agent["x_agents_core"], "harness_config": {key: value}} + reject("codex_hidden_config_" + key + "_" + str(value), {"agent": {**agent, "x_agents_core": extension}}, "harness_config contains unsupported or invalid native model parameters") + for verbosity in (None, "medium", "low", "high"): + options = dict(agent) + if verbosity is not None: + options["text"] = {"verbosity": verbosity} + proof["current_case"] = {"case": "codex_verbosity_" + (verbosity or "default"), "step": "session_admission", "verbosity": verbosity or "medium"} + record(proof) + session = sessions.create(agent=options, **session_options) + cleanup.callback(delete_session, sessions, session.id) + run(session, "codex_verbosity_" + (verbosity or "default"), "What is two plus two? Answer with exactly 4 and do not use tools.", "4", verbosity or "medium") + options = {**agent, "x_agents_core": {**agent["x_agents_core"], "harness_config": {"model_reasoning_effort": "medium"}}} + session = sessions.create(agent=options, **session_options) + cleanup.callback(delete_session, sessions, session.id) + assert session.agent.to_dict()["x_agents_core"]["harness_config"] == {"model_reasoning_effort": "medium"} + run(session, "codex_valid_reasoning_effort", "What is two plus two? Answer with exactly 4 and do not use tools.", "4") + proof["passed"] = True + return proof["checks"] + finally: + record(proof) diff --git a/services/core/tests/official_native_steering.py b/services/core/tests/official_native_steering.py new file mode 100644 index 000000000..0f08508fe --- /dev/null +++ b/services/core/tests/official_native_steering.py @@ -0,0 +1,140 @@ +"""MiniMax active-message delivery through the public API and native workspace.""" + +import shlex +import time +import uuid + +from session_cleanup import delete_session + + +def verify_native_steering(client, foreign, http, agent_options, session_options, ready, record): + assert agent_options["x_agents_core"]["harness"] == "mcode", "Select the MiniMax Harness" + assert session_options["environment"]["type"] != "none", "A native workspace is required" + sessions = client.beta.agents.sessions + files = client.beta.agents.environments.files + workspace = session_options["environment"].get("workspace_directory", "/workspace").rstrip("/") + nonce = uuid.uuid4().hex + marker = "steering-started-" + nonce + memory = "steering-memory-" + uuid.uuid4().hex + command = "printf start >> " + shlex.quote(workspace + "/" + marker) + "; sleep 45" + initial = ("Run this command exactly once with your native shell tool in the foreground. " + "Wait until it finishes before replying. Do not background, delegate, retry or write other files.\n" + command) + followup = "Remember this conversation-only token for our next turn: " + memory + ". Keep waiting for the foreground command; do not write this token to a file." + proof = {"harness": agent_options["x_agents_core"]["harness"], "model": agent_options["model"], + "checks": [], "runs": [], "passed": False, + "unverified": ["Direct ACP receipt and native duplicate mode are not observed by public HTTP retries.", + "Native acceptance does not guarantee model consumption of the active input."], + "initial_input": initial, "active_input": followup} + session = sessions.create(agent=agent_options, **session_options) + endpoint = str(client.base_url).rstrip("/") + "/agents/sessions/" + session.id + headers = {"Authorization": "Bearer " + client.api_key, "OpenAI-Beta": "agents=v1"} + proof.update(session=session.id, environment=session.environment.id) + + def message(text): + return {"type": "agent.session.input.message", "input": [{"role": "user", "content": [{"type": "input_text", "text": text}]}]} + + def items(): + response = http.get(endpoint + "/items", headers=headers, params={"limit": 100, "order": "asc"}) + assert response.status_code == 200 and not response.json()["has_more"] + raw = response.json()["data"] + assert raw == [item.to_dict() for item in sessions.items.list(session.id, limit=100, order="asc").data] + return raw + + def collect(stream, events): + for event in stream: + events.append(event.to_dict()) + assert event.type not in {"error", "agent.session.failed", "agent.session.turn.failed", "agent.session.turn.cancelled", "agent.session.requires_action"} + if event.type == "agent.session.idle" and any(e["type"] == "agent.session.turn.created" for e in events): + break + else: + raise AssertionError("SSE ended without a completed Turn and idle Session") + created = [e for e in events if e["type"] == "agent.session.turn.created"] + terminal = [e for e in events if e["type"] == "agent.session.turn.completed"] + assert len(created) == len(terminal) == 1 + turn = terminal[0]["turn"]["id"] + assert created[0]["turn"]["id"] == turn + assert events.index(created[0]) < events.index(terminal[0]) < len(events) - 1 + assert len({e["event_id"] for e in events}) == len(events) + assert all(str(uuid.UUID(e["event_id"])) == e["event_id"] for e in events) + assert all(e.get("turn_id") == turn for e in events if e["type"].startswith("agent.session.turn.")) + assert sessions.retrieve(session.id).status == "idle" and not sessions.retrieve(session.id).required_actions + return turn + + try: + ready(session) + observed = [] + proof["runs"].append(observed) + with sessions.events.stream(session.id, timeout=300) as stream: + sessions.events.create(session.id, events=[message(initial)], idempotency_key="initial-" + nonce) + deadline = time.monotonic() + 180 + while time.monotonic() < deadline: + turns = list(sessions.turns.list(session.id, limit=100)) + assert len(turns) <= 1 and all(t.status not in {"completed", "failed", "cancelled"} for t in turns), "Native Turn ended before active input" + markers = [f for f in files.list(session.environment.id, path="/workspace") if f.path == "/workspace/" + marker] + if markers: + assert len(markers) == 1 and markers[0].size_bytes == 5, "Native foreground command was repeated" + assert len(turns) == 1 and turns[0].status == "in_progress" + active_turn = turns[0].id + break + time.sleep(0.2) + else: + raise AssertionError("Native started marker was not observed") + assert sessions.turns.retrieve(active_turn, session_id=session.id).status == "in_progress" + proof["barrier"] = {"path": "/workspace/" + marker, "size_bytes": 5, "turn_id": active_turn} + key = "active-" + nonce + response = sessions.events.with_raw_response.create(session.id, events=[message(followup)], idempotency_key=key) + assert response.status_code == 202 and response.content == b"" and response.parse() is None + assert sessions.turns.retrieve(active_turn, session_id=session.id).status == "in_progress", "Turn settled during active-input admission" + proof["admission"] = {"status": 202, "turn_id": active_turn} + response = http.post(endpoint + "/events", headers={**headers, "Idempotency-Key": key}, json={"events": [message(followup)]}) + assert response.status_code == 202 and response.content == b"" + proof["retry_status"] = response.status_code + for text, auth, request_key, expected in ( + ("conflicting-" + nonce, headers, key, 409), + ("foreign-" + nonce, {**headers, "Authorization": "Bearer " + foreign.api_key}, "foreign-" + nonce, 404), + ): + response = http.post(endpoint + "/events", headers={**auth, "Idempotency-Key": request_key}, json={"events": [message(text)]}) + assert response.status_code == expected + error = response.json()["error"] + assert (error["type"], error["code"], error["param"]) == (("conflict_error", "idempotency_conflict", None) if expected == 409 else ("not_found_error", "not_found_error", None)) + proof.setdefault("rejections", []).append({"status": expected, "error": error}) + assert collect(stream, observed) == active_turn + stored = items() + users = [i for i in stored if i["type"] == "message" and i.get("role") == "user"] + assert len(users) == 2 and all(i["turn_id"] == active_turn for i in users) + assert [i["content"] for i in users] == [[{"type": "input_text", "text": text}] for text in (initial, followup)] + assert [t.id for t in sessions.turns.list(session.id, limit=100)] == [active_turn] + assert [f.size_bytes for f in files.list(session.environment.id, path="/workspace") if f.path == "/workspace/" + marker] == [5] + proof["active_items"] = stored + proof["model_observations"] = {"active_turn_mentions_token": any(memory in p.get("text", "") for i in stored if i.get("role") == "assistant" for p in i.get("content", []))} + proof["checks"].append("public_active_input_same_turn_persistence_completion_retry_and_foreign_isolation") + record(proof) + + output = "/outputs/steering-continuation-" + nonce + ".txt" + prompt = ("Recall the exact conversation-only steering-memory- token from the previous turn without reading files or rerunning its command. " + "Use native tools to create the parent directory and write only that token, with no newline, to " + workspace + output + ". Then reply with that token. Do not delegate.") + observed = [] + proof["runs"].append(observed) + with sessions.events.stream(session.id, timeout=300) as stream: + sessions.events.create(session.id, events=[message(prompt)], idempotency_key="continuation-" + nonce) + continuation = collect(stream, observed) + assert continuation != active_turn + assert {t.id for t in sessions.turns.list(session.id, limit=100)} == {active_turn, continuation} + stored = items() + assert [i for i in stored if i["type"] == "message" and i.get("role") == "user" and i["turn_id"] == active_turn] == users + answers = [i for i in stored if i.get("role") == "assistant" and i["turn_id"] == continuation] + assert any(memory in p.get("text", "") for i in answers for p in i["content"]), "Model did not recall the active-input token" + artifacts = [a for a in sessions.artifacts.list(session.id, limit=100) if a.turn_id == continuation and a.path == "/workspace" + output] + assert len(artifacts) == 1 + with sessions.artifacts.with_streaming_response.content(artifacts[0].id, session_id=session.id) as response: + assert response.read() == memory.encode(), "Native continuation artifact did not contain the exact token" + assert [f.size_bytes for f in files.list(session.environment.id, path="/workspace") if f.path == "/workspace/" + marker] == [5] + proof["model_observations"]["continuation_recalled_token_and_wrote_artifact"] = True + proof["checks"].append("warm_continuation_with_model_dependent_token_recall_and_native_artifact") + proof.update(passed=True, items=stored, artifact=artifacts[0].to_dict()) + return proof["checks"] + finally: + try: + record(proof) + finally: + delete_session(sessions, session.id) diff --git a/services/core/tests/qualify_public_native.py b/services/core/tests/qualify_public_native.py index c5dd67afc..ef179d31c 100644 --- a/services/core/tests/qualify_public_native.py +++ b/services/core/tests/qualify_public_native.py @@ -17,6 +17,8 @@ from official_hosted_functions_native import verify_hosted_functions from official_hosted_structured_native import verify_hosted_structured +from official_native_policies import verify_native_policies +from official_native_steering import verify_native_steering from official_pending_actions_native import verify_pending_actions from official_workspace_images_native import verify_workspace_images from official_environment_composition import verify_composition @@ -79,13 +81,13 @@ def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--settings", required=True, help="Private JSON with agent, model_provider and environment") parser.add_argument("--foreign-key-file", required=True, help="Private key of a different Project") - parser.add_argument("--suite", required=True, choices=("none", "functions", "pending-actions", "images", "structured", "composition")) + parser.add_argument("--suite", required=True, choices=("none", "functions", "tool-search", "policies", "steering", "pending-actions", "images", "structured", "composition")) parser.add_argument("--evidence", required=True, type=Path, help="New evidence file under ~/.oac") parser.add_argument("--compose-directory", type=Path, help="Owned installation to restart for cold recovery") parser.add_argument("--compose-project", help="Exact owned Compose project; required with --compose-directory") args = parser.parse_args() assert bool(args.compose_directory) == bool(args.compose_project), "Supply both Compose selectors or neither" - assert args.suite != "pending-actions" or args.compose_directory is None, "Pending actions tests client reconnect, not process restart" + assert args.suite not in {"pending-actions", "policies", "steering"} or args.compose_directory is None, "This suite does not qualify process restart" evidence = args.evidence assert evidence.is_absolute() and not evidence.exists(), "Use a new absolute evidence path" assert evidence.parent.resolve().is_relative_to((Path.home() / ".oac").resolve()), "Evidence belongs under ~/.oac" @@ -167,12 +169,15 @@ def ready(session): raise AssertionError("Environment did not connect; use a separate sandbox for each Session") session_options = {"environment": environment, "extra_body": {"x_agents_core": {"model_provider": provider}}} - suites = {"none": verify_none, "functions": verify_hosted_functions, "pending-actions": verify_pending_actions, + suites = {"none": verify_none, "functions": verify_hosted_functions, "tool-search": verify_hosted_functions, + "policies": verify_native_policies, "steering": verify_native_steering, "pending-actions": verify_pending_actions, "images": verify_workspace_images, "structured": verify_hosted_structured, "composition": verify_composition} try: kwargs = {"ready": ready, "record": record} - if args.suite != "pending-actions": + if args.suite not in {"pending-actions", "policies", "steering"}: kwargs["restart"] = restart + if args.suite == "tool-search": + kwargs["deferred"] = True checks = suites[args.suite](client, foreign, http, agent, session_options, **kwargs) report.update(status="passed", checks=checks) if restart is not None: diff --git a/services/core/tests/qualify_public_native_test.py b/services/core/tests/qualify_public_native_test.py index 5612da39c..371450897 100644 --- a/services/core/tests/qualify_public_native_test.py +++ b/services/core/tests/qualify_public_native_test.py @@ -69,6 +69,35 @@ def suite(*args, record, **kwargs): self.run_suite(suite) self.assertFalse(self.evidence.exists()) + def test_active_policy_suites_cannot_claim_process_restart(self): + for name in ("policies", "steering"): + with self.subTest(suite=name): + argv = [name if value == "none" else value for value in self.argv] + argv += ["--compose-directory", str(self.root), "--compose-project", "owned-qualification"] + with patch("sys.argv", argv), patch.object(qualification.subprocess, "run") as run: + with self.assertRaisesRegex(AssertionError, "does not qualify process restart"): + qualification.main() + run.assert_not_called() + + def test_deferred_variant_is_explicit_and_policy_failures_do_not_pass(self): + self.settings["environment"] = {"type": "openai_hosted"} + (self.root / "settings.json").write_text(json.dumps(self.settings)) + for name, target in (("tool-search", "verify_hosted_functions"), ("policies", "verify_native_policies"), ("steering", "verify_native_steering")): + with self.subTest(suite=name): + argv = [name if value == "none" else value for value in self.argv] + def suite(*args, **kwargs): + self.assertEqual(kwargs.get("deferred", False), name == "tool-search") + self.assertEqual("restart" in kwargs, name == "tool-search") + kwargs["record"]({"unverified": ["native policy behavior"]}) + raise AssertionError("native admission blocked") + with patch("sys.argv", argv), patch.object(qualification, target, suite): + with self.assertRaisesRegex(AssertionError, "native admission blocked"): + qualification.main() + proof = json.loads(self.evidence.read_text()) + self.assertEqual((proof["status"], proof["cold_recovery"]), ("failed", "unverified")) + self.assertEqual(proof["proof"]["unverified"], ["native policy behavior"]) + self.evidence.unlink() + def test_custom_workspace_uses_physical_tools_and_logical_public_file_paths(self): client = MagicMock() session = SimpleNamespace(status="idle", required_actions=[], environment=SimpleNamespace( @@ -110,6 +139,20 @@ def reject(request): qualification.verify_hosted_structured(client, client, http, self.settings["agent"], options, ready=lambda session: None, restart=None, record=lambda proof: None) self.assertEqual(bodies[-1]["x_agents_core"], self.settings["agent"]["x_agents_core"]) + selected = {**self.settings["agent"], "x_agents_core": {"harness": "claude_sdk"}} + with self.assertRaises(BadRequestError): + qualification.verify_hosted_functions(client, client, http, selected, options, + ready=lambda session: None, restart=None, record=lambda proof: None, deferred=True) + tools = bodies[-1]["agent"]["tools"] + self.assertEqual(sum(tool["type"] == "tool_search" for tool in tools), 1) + self.assertTrue(next(tool for tool in tools if tool["type"] == "function")["defer_loading"]) + self.assertEqual(bodies[-1]["x_agents_core"]["model_provider"], self.settings["model_provider"]) + selected["x_agents_core"] = {"harness": "mcode"} + with self.assertRaises(BadRequestError): + qualification.verify_native_steering(client, client, http, selected, options, + ready=lambda session: None, record=lambda proof: None) + self.assertEqual(bodies[-1]["environment"], options["environment"]) + self.assertEqual(bodies[-1]["agent"], selected) def test_restart_requires_a_running_host_and_observed_new_process(self): (self.root / "compose.yaml").write_text("services: {}\n")