4203086899
Upgrade from mock-only to REAL multi-backend validation:
Backends (skillopt/sleep/backend.py):
- CliBackend base: shared attempt/judge/reflect prompts, response cache,
token accounting. Subclasses implement only _call().
- ClaudeCliBackend: drives `claude -p --output-format text`.
- CodexCliBackend: drives the REAL @openai/codex `exec -o <file>` for clean
output; resolve_codex_path() skips the hermes wrapper at ~/.local/bin/codex.
- reflect() now aggregates the exact failing judge criteria into the prompt
(gbrain's lesson: tell the optimizer what the scorer rewards).
Rule judges (skillopt/sleep/judges.py): gbrain-compatible local scorers
(section_present / regex / max_chars / contains / tool_called) — held-out
scoring with no judge-API spend. TaskRecord gains a `judge` field +
reference_kind="rule".
gbrain-evals adapter (experiments/gbrain_bench.py, run_gbrain.py): load
garrytan/gbrain-evals skillopt-v1 deficient skills + train/held-out task
sets and run our consolidate() loop against the SAME suite gbrain scores.
REAL results (docs/sleep/real_api_results.md), brief-writer seed, 1 night:
- Claude (Haiku): held-out 0.00 -> 1.00
- Codex: held-out 0.00 -> 0.67
Both proposed a correct, general format rule into the protected LEARNED block.
CLI: --backend {mock,claude,codex}, --codex-path, --model; experiment +
gbrain runners gain --limit-* cost controls. 17 tests pass.
Co-Authored-By: Claude Opus 4 <noreply@anthropic.com>
145 lines
5.8 KiB
Python
145 lines
5.8 KiB
Python
"""SkillOpt-Sleep — run the gbrain-evals skillopt-v1 benchmark with our engine.
|
|
|
|
Reproduces gbrain's "Result 1 — skills measurably improve" scorecard
|
|
(docs/benchmarks/2026-06-03-skillopt.md) using SkillOpt-Sleep's
|
|
consolidate() loop and either the claude or codex backend.
|
|
|
|
For each deficient seed skill:
|
|
1. score the held-out tasks with the ORIGINAL skill -> before
|
|
2. run N consolidation nights on the training tasks (gated) -> evolve skill
|
|
3. score the held-out tasks with the EVOLVED skill -> after
|
|
|
|
Held-out scoring is done locally by the rule judge (no judge API). Only the
|
|
agent's `attempt` (and the optimizer's `reflect`) spend tokens.
|
|
|
|
Usage:
|
|
python -m skillopt.sleep.experiments.run_gbrain --backend mock
|
|
python -m skillopt.sleep.experiments.run_gbrain --backend claude --seeds brief-writer --nights 2
|
|
python -m skillopt.sleep.experiments.run_gbrain --backend codex --data-root /tmp/gbrain-evals/eval/data/skillopt-v1
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import sys
|
|
from typing import Dict, List, Optional
|
|
|
|
from skillopt.sleep.backend import get_backend
|
|
from skillopt.sleep.consolidate import consolidate, select_gate_score
|
|
from skillopt.sleep.experiments.gbrain_bench import (
|
|
available_seeds,
|
|
find_data_root,
|
|
load_seed,
|
|
)
|
|
from skillopt.sleep.replay import aggregate_scores, replay_batch
|
|
|
|
|
|
def _score(backend, tasks, skill, memory, split="holdout", metric="mixed", w=0.5):
|
|
sub = [t for t in tasks if t.split == split] or tasks
|
|
pairs = replay_batch(backend, sub, skill, memory)
|
|
h, s = aggregate_scores(pairs)
|
|
return h, s, select_gate_score(h, s, metric, w)
|
|
|
|
|
|
def run_seed(backend, seed: str, skill: str, tasks: List, *,
|
|
nights: int = 3, edit_budget: int = 4,
|
|
limit_replay: int = 0, limit_holdout: int = 0) -> dict:
|
|
memory = ""
|
|
# optionally cap each split to control API cost / latency
|
|
if limit_replay or limit_holdout:
|
|
replay = [t for t in tasks if t.split == "replay"]
|
|
holdout = [t for t in tasks if t.split == "holdout"]
|
|
if limit_replay:
|
|
replay = replay[:limit_replay]
|
|
if limit_holdout:
|
|
holdout = holdout[:limit_holdout]
|
|
tasks = replay + holdout
|
|
bh, bs, bscore = _score(backend, tasks, skill, memory)
|
|
trace = [{"night": 0, "held_out_hard": round(bh, 3), "action": "baseline"}]
|
|
cur = skill
|
|
for night in range(1, nights + 1):
|
|
res = consolidate(
|
|
backend, tasks, cur, memory,
|
|
edit_budget=edit_budget, gate_metric="mixed", gate_mixed_weight=0.5,
|
|
evolve_skill=True, evolve_memory=False, night=night,
|
|
)
|
|
if res.accepted:
|
|
cur = res.new_skill
|
|
trace.append({
|
|
"night": night,
|
|
"held_out_hard": round(res.holdout_candidate, 3),
|
|
"action": res.gate_action,
|
|
"accepted": res.accepted,
|
|
"edits": [e.content for e in res.applied_edits],
|
|
})
|
|
if res.holdout_candidate >= 0.999:
|
|
break
|
|
ah, as_, ascore = _score(backend, tasks, cur, memory)
|
|
return {
|
|
"seed": seed,
|
|
"held_out_before": round(bh, 3),
|
|
"held_out_after": round(ah, 3),
|
|
"improved": ah > bh,
|
|
"nights": len(trace) - 1,
|
|
"trace": trace,
|
|
"final_skill_tail": cur[-400:],
|
|
}
|
|
|
|
|
|
def main(argv=None) -> int:
|
|
ap = argparse.ArgumentParser(description="Run gbrain-evals skillopt-v1 with SkillOpt-Sleep")
|
|
ap.add_argument("--backend", default="mock", choices=["mock", "claude", "codex"])
|
|
ap.add_argument("--model", default="")
|
|
ap.add_argument("--codex-path", default="")
|
|
ap.add_argument("--data-root", default="", help="path to eval/data/skillopt-v1")
|
|
ap.add_argument("--seeds", default="", help="comma list; default = all available")
|
|
ap.add_argument("--nights", type=int, default=3)
|
|
ap.add_argument("--edit-budget", type=int, default=4)
|
|
ap.add_argument("--limit-replay", type=int, default=0, help="cap #training tasks (cost control)")
|
|
ap.add_argument("--limit-holdout", type=int, default=0, help="cap #held-out tasks (cost control)")
|
|
ap.add_argument("--json", action="store_true")
|
|
args = ap.parse_args(argv)
|
|
|
|
data_root = find_data_root(args.data_root)
|
|
if not data_root:
|
|
print("ERROR: could not find eval/data/skillopt-v1. Clone gbrain-evals and pass --data-root.",
|
|
file=sys.stderr)
|
|
return 2
|
|
|
|
seeds = [s.strip() for s in args.seeds.split(",") if s.strip()] or available_seeds(data_root)
|
|
backend = get_backend(args.backend, model=args.model, codex_path=args.codex_path)
|
|
|
|
results = []
|
|
for seed in seeds:
|
|
skill, tasks = load_seed(data_root, seed)
|
|
if not tasks:
|
|
continue
|
|
r = run_seed(backend, seed, skill, tasks, nights=args.nights,
|
|
edit_budget=args.edit_budget,
|
|
limit_replay=args.limit_replay, limit_holdout=args.limit_holdout)
|
|
results.append(r)
|
|
if not args.json:
|
|
print(f" {seed:<18} held-out {r['held_out_before']:.2f} -> {r['held_out_after']:.2f}"
|
|
f" ({'IMPROVED' if r['improved'] else 'no change'}, {r['nights']} nights)")
|
|
|
|
n_improved = sum(1 for r in results if r["improved"])
|
|
summary = {
|
|
"benchmark": "gbrain-evals/skillopt-v1",
|
|
"backend": backend.name,
|
|
"model": args.model or "(default)",
|
|
"n_seeds": len(results),
|
|
"n_improved": n_improved,
|
|
"tokens_used": backend.tokens_used(),
|
|
"results": results,
|
|
}
|
|
if args.json:
|
|
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
|
else:
|
|
print(f"\n=== {n_improved}/{len(results)} seeds improved on held-out "
|
|
f"(backend={backend.name}, ~{backend.tokens_used()} tokens) ===")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|