Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 14 additions & 3 deletions skillopt_sleep/backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,7 @@
from __future__ import annotations

import json
import math
import os
import re
import secrets
Expand Down Expand Up @@ -138,6 +139,8 @@ def _optimizer_feedback(task: TaskRecord, result: ReplayResult) -> str:
task.judge,
getattr(result, "response", ""),
getattr(result, "tools_called", []),
# Replay measured tool calls for every tool task.
verified_tools=True,
)
return feedback
feedback = getattr(result, "optimizer_feedback", "")
Expand Down Expand Up @@ -448,11 +451,19 @@ def judge(self, task: TaskRecord, response: str) -> Tuple[float, float, str]:
raw = self._cached_call(key, prompt, max_tokens=200)
obj = _extract_json(raw, "object")
if isinstance(obj, dict):
raw_score = obj.get("score", 0.0)
try:
soft = float(obj.get("score", 0.0))
return (1.0 if soft >= 0.8 else 0.0), soft, str(obj.get("reason", ""))[:200]
soft = float(raw_score)
except (ValueError, TypeError):
pass
soft = None
# The judge prompt asks for a 0..1 score. json.loads accepts NaN and
# Infinity, and a model may answer on another scale (0-10, percent).
# Clamping would turn "8/10" into a perfect 1.0, so fail closed and
# name the problem instead of letting it move the gate.
if soft is not None and not isinstance(raw_score, bool):
if math.isfinite(soft) and 0.0 <= soft <= 1.0:
return (1.0 if soft >= 0.8 else 0.0), soft, str(obj.get("reason", ""))[:200]
return 0.0, 0.0, f"judge-score-out-of-range: {str(raw_score)[:40]}"
return 0.0, 0.0, "judge-parse-failed"

def reflect(
Expand Down
21 changes: 19 additions & 2 deletions skillopt_sleep/consolidate.py
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,19 @@ class ConsolidationResult:
gate_trials: List[dict] = field(default_factory=list)


def task_content_key(task: TaskRecord) -> Tuple[str, str]:
"""What a task asks, independent of its id.

Task ids hash the project with the intent and differ between miners, so
two records of the same request can carry different ids. The key is the
case- and whitespace-normalized intent plus context excerpt.
"""
def _norm(text: str) -> str:
return " ".join(str(text or "").lower().split())

return _norm(task.intent), _norm(task.context_excerpt)


def _split(tasks: List[TaskRecord]) -> Tuple[List[TaskRecord], List[TaskRecord], bool]:
"""Return ``(train_tasks, val_tasks, holdout_leaked)``.

Expand All @@ -69,7 +82,10 @@ def _split(tasks: List[TaskRecord]) -> Tuple[List[TaskRecord], List[TaskRecord],
(replay->train, holdout->val) for robustness.

``holdout_leaked`` is True when val is not disjoint from train — i.e. the
gate would score the very tasks the edits were derived from. A single mined
gate would score the very tasks the edits were derived from. Disjointness
is checked by id AND by :func:`task_content_key`, so a val task's twin
under another id (for example one recalled from another project's archive)
also counts as a leak. A single mined
task that carries a train/val (or legacy) split always lands here; a lone
``test`` task instead yields empty train/val and no gate, so it is not
flagged as leaked. Such a non-disjoint comparison cannot detect overfitting,
Expand All @@ -94,7 +110,8 @@ def _norm(s: str) -> str:
leaked = leaked or bool(train)
if not leaked and train and val:
train_ids = {t.id for t in train}
if any(t.id in train_ids for t in val):
train_keys = {task_content_key(t) for t in train}
if any(t.id in train_ids or task_content_key(t) in train_keys for t in val):
leaked = True
return train, val, leaked

Expand Down
28 changes: 21 additions & 7 deletions skillopt_sleep/dream.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,11 @@
import re
from typing import List, Optional

from skillopt_sleep.consolidate import ConsolidationResult, consolidate
from skillopt_sleep.consolidate import (
ConsolidationResult,
consolidate,
task_content_key,
)
from skillopt_sleep.types import TaskRecord

# ── synthetic augmentation ("dream up" variants of today's tasks) ─────────────
Expand Down Expand Up @@ -69,26 +73,31 @@ def recall_similar(
k: int,
*,
exclude_ids: Optional[set[str]] = None,
exclude_tasks: Optional[List[TaskRecord]] = None,
) -> List[TaskRecord]:
"""Return the ``k`` historical tasks most lexically similar to any of
tonight's ``new_tasks`` (max Jaccard token overlap). Recalled tasks are
returned as training material (split='train'); deterministic, stdlib-only.

Archived val/test tasks are never recalled, and ``exclude_ids`` blocks
tonight's held-out ids (and their ``derived_from`` sources) from re-entering
the training pool.
the training pool. Ids alone are not enough: the archive is shared across
projects and ids hash the project with the intent, so ``exclude_tasks``
also blocks any archived task whose :func:`task_content_key` matches one of
the given (held-out) tasks.
"""
if not history or k <= 0 or not new_tasks:
return []
blocked = set(exclude_ids or ())
blocked_keys = {task_content_key(t) for t in (exclude_tasks or ())}
for t in new_tasks:
blocked.add(t.id)
if t.derived_from:
blocked.add(t.derived_from)
new_tok = [_tokens(t.intent) for t in new_tasks]
scored = []
for h in history:
if h.id in blocked:
if h.id in blocked or task_content_key(h) in blocked_keys:
continue
if _normalize_split(h.split) in ("val", "test"):
continue
Expand Down Expand Up @@ -146,14 +155,19 @@ def dream_consolidate(
train = [t for t in tasks if t.split == "train"]
enlarged = list(tasks)
if recall_k > 0 and history_tasks:
held_out_ids = {
t.id for t in tasks if _normalize_split(t.split) in ("val", "test")
}
# Block held-out tasks by id AND by content: the archive is shared
# across projects, so tonight's val task can be archived under another
# project's id.
held_out = [
t for t in tasks if _normalize_split(t.split) in ("val", "test")
]
held_out_ids = {t.id for t in held_out}
for t in tasks:
if t.derived_from:
held_out_ids.add(t.derived_from)
enlarged += recall_similar(
train, history_tasks, recall_k, exclude_ids=held_out_ids,
train, history_tasks, recall_k,
exclude_ids=held_out_ids, exclude_tasks=held_out,
)
if dream_factor > 0:
seed = [t for t in enlarged if t.split == "train" and t.origin != "dream"]
Expand Down
33 changes: 32 additions & 1 deletion skillopt_sleep/harvest.py
Original file line number Diff line number Diff line change
Expand Up @@ -177,6 +177,35 @@ def _is_meta_prompt(text: str) -> bool:
"## Skill\n",
)

# Shared by the tool-enabled attempt prompts of the Claude, Codex, Copilot and
# OpenCode backends (matched case-insensitively).
_TOOL_ATTEMPT_MARKER = "treat a 'learned preferences' block as"


def _engine_prompt_markers() -> tuple:
"""Opening lines of the engine's own prompt templates.

The static markers above were written for earlier prompt wording and no
longer match the current registry, so replay sessions slipped through.
Deriving markers from the registry (defaults and any active override)
keeps this filter in step with the prompts the engine actually sends.
"""
from skillopt_sleep import prompts

markers = []
for name, meta in prompts.DEFAULTS.items():
texts = [meta["text"]]
try:
texts.append(prompts.get_prompt(name))
except Exception:
pass # an unreadable override file must not break harvesting
for text in texts:
head = str(text).split("__", 1)[0].strip().splitlines()
# Short openings are too generic to identify an engine prompt.
if head and len(head[0].strip()) >= 24:
markers.append(head[0].strip())
return tuple(dict.fromkeys(markers))


# Sessions written by OTHER tools' sub-agents (memory observers, critic
# sub-agents, plugin self-invocations). These are multi-turn, so the
Expand Down Expand Up @@ -217,9 +246,11 @@ def _is_headless_replay(digest: "SessionDigest") -> bool:
if digest.n_user_turns == 0:
return True
prompt = digest.user_prompts[0] if digest.user_prompts else ""
for marker in _REPLAY_PROMPT_MARKERS:
for marker in _REPLAY_PROMPT_MARKERS + _engine_prompt_markers():
if marker in prompt:
return True
if _TOOL_ATTEMPT_MARKER in prompt.lower():
return True
# Sub-3-second single-turn sessions with short prompts are almost
# certainly programmatic (engine grader/judge calls). We require the
# prompt to also be short (<200 chars) to avoid false-positives on
Expand Down
29 changes: 24 additions & 5 deletions skillopt_sleep/judges.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,9 @@
* no_refusal — the response is not a bare refusal/abstention
* tool_called <name> — a tool with <name> was invoked (needs a tool loop;
in single-shot replay we approximate via an
explicit "TOOL_CALL: <name>" marker the agent emits)
explicit "TOOL_CALL: <name>" marker the agent emits;
once tool calls were measured, only the measured
calls count -- see ``verified_tools``)

Ops divide into two families, and the distinction is load-bearing:

Expand Down Expand Up @@ -100,12 +102,16 @@ def _is_refusal(response: str) -> bool:


def _check(op: str, arg: Any, response: str,
tools_called: List[str]) -> Tuple[bool, str]:
tools_called: List[str], verified_tools: bool = False) -> Tuple[bool, str]:
"""Evaluate one check.

Returns ``(passed, problem)``. ``problem`` is non-empty only when the check
itself is malformed (e.g. an unparseable regex) rather than simply unmet —
the two need opposite fixes, so they must not look alike in the rationale.

``verified_tools`` means ``tools_called`` was measured by a tool route, so
a ``TOOL_CALL:`` marker in the response text is a self-report, not
evidence, and must not satisfy ``tool_called``.
"""
r = response or ""
if op == "section_present":
Expand Down Expand Up @@ -134,6 +140,8 @@ def _check(op: str, arg: Any, response: str,
name = str(arg).lower()
if any(name == t.lower() for t in tools_called):
return True, ""
if verified_tools:
return False, ""
# single-shot approximation: the agent emits an explicit marker
return bool(re.search(r"(?i)\btool_call\s*:\s*%s\b" % re.escape(name), r)), ""
# unknown op: do not block
Expand Down Expand Up @@ -301,8 +309,15 @@ def score_rule_judge_with_feedback(
judge: Dict[str, Any],
response: str,
tools_called: List[str] | None = None,
*,
verified_tools: bool = False,
) -> Tuple[float, float, str, str]:
"""Return scores, audit rationale, and optimizer-safe feedback."""
"""Return scores, audit rationale, and optimizer-safe feedback.

Pass ``verified_tools=True`` when ``tools_called`` came from a tool route
(``attempt_with_tools``); the response's own ``TOOL_CALL:`` markers are
then ignored for ``tool_called`` checks.
"""
checks = (judge or {}).get("checks", []) or []
if not checks:
return (
Expand All @@ -316,7 +331,9 @@ def score_rule_judge_with_feedback(
failed_desc: List[str] = []
semantic_failures: List[str] = []
for c in checks:
ok, problem = _check(c.get("op", ""), c.get("arg"), response, tools_called)
ok, problem = _check(
c.get("op", ""), c.get("arg"), response, tools_called, verified_tools
)
if ok:
passed += 1
else:
Expand All @@ -336,9 +353,11 @@ def score_rule_judge(
judge: Dict[str, Any],
response: str,
tools_called: List[str] | None = None,
*,
verified_tools: bool = False,
) -> Tuple[float, float, str]:
"""Return the backward-compatible (hard, soft, rationale) tuple."""
hard, soft, rationale, _feedback = score_rule_judge_with_feedback(
judge, response, tools_called
judge, response, tools_called, verified_tools=verified_tools
)
return hard, soft, rationale
37 changes: 31 additions & 6 deletions skillopt_sleep/memory.py
Original file line number Diff line number Diff line change
Expand Up @@ -44,10 +44,25 @@ def _strip_learned(doc: str) -> str:
return doc.rstrip()


def _learned_item(text: str) -> str:
"""Normalize one learned item to the single bullet line it is stored as.

The block is read back one ``- `` line per item, so an item must not span
lines: internal whitespace (including newlines) collapses to single
spaces. Exactly one leading bullet marker is dropped, so ``- X`` and ``X``
are the same item while text such as ``--force`` keeps its dashes.
"""
item = " ".join(str(text or "").split())
if item.startswith("- "):
item = item[2:].lstrip()
return item


def set_learned(doc: str, learned_lines: List[str]) -> str:
"""Replace the protected learned region with the given bullet lines."""
base = _strip_learned(doc)
body = "\n".join(f"- {ln.strip().lstrip('- ').strip()}" for ln in learned_lines if ln.strip())
items = (_learned_item(ln) for ln in learned_lines)
body = "\n".join(f"- {item}" for item in items if item)
block = (
f"\n\n{LEARNED_START}\n"
f"## Learned preferences & procedures\n\n{_BANNER}\n\n{body}\n"
Expand All @@ -57,12 +72,21 @@ def set_learned(doc: str, learned_lines: List[str]) -> str:


def current_learned_lines(doc: str) -> List[str]:
"""Return the learned items, one per ``- `` bullet.

Blocks written before items were normalized to one line can contain an
item whose continuation lines do not start with ``- ``. Those lines are
joined onto the preceding item instead of being dropped, so adopted text
survives the next edit.
"""
inner = extract_learned(doc)
lines: List[str] = []
for ln in inner.splitlines():
ln = ln.strip()
if ln.startswith("- "):
lines.append(ln[2:].strip())
lines.append(_learned_item(ln))
elif ln and lines:
lines[-1] = _learned_item(f"{lines[-1]} {ln}")
return lines


Expand Down Expand Up @@ -110,11 +134,12 @@ def apply_edits_detailed(
for e in edits:
op = (e.op or "add").lower()
if op == "add":
if _norm(e.content) in norm_set or not e.content.strip():
item = _learned_item(e.content)
if not item or _norm(item) in norm_set:
unmatched.append(e)
continue
lines.append(e.content.strip())
norm_set.add(_norm(e.content))
lines.append(item)
norm_set.add(_norm(item))
applied.append(e)
elif op == "delete":
anchor = _norm(e.anchor or e.content)
Expand All @@ -130,7 +155,7 @@ def apply_edits_detailed(
unmatched.append(e)
elif op == "replace":
anchor = _norm(e.anchor)
replacement = e.content.strip()
replacement = _learned_item(e.content)
new_lines = []
changed = False
for line in lines:
Expand Down
5 changes: 4 additions & 1 deletion skillopt_sleep/replay.py
Original file line number Diff line number Diff line change
Expand Up @@ -50,8 +50,11 @@ def replay_one(backend: Backend, task: TaskRecord, skill: str, memory: str,
# rule judges may need the detected tool calls; score locally when possible
if task.reference_kind == "rule" and task.judge:
from skillopt_sleep.judges import score_rule_judge_with_feedback
# Tool tasks went through attempt_with_tools, which measured the calls
# (the default marker fallback converts markers there), so the
# response's own TOOL_CALL text is not evidence of a call.
hard, soft, rationale, optimizer_feedback = score_rule_judge_with_feedback(
task.judge, response, tools_called
task.judge, response, tools_called, verified_tools=bool(tools)
)
else:
hard, soft, rationale = backend.judge(task, response)
Expand Down
Loading