Robustness for the claude/codex backends on Windows: argv overflow, subprocess encoding, tolerant JSON, test-eval dirs
Fixes surfaced running SkillOpt end-to-end on the bundled `claude` backend
(local Claude CLI) on Windows. None changes the OpenAI/GPT happy path.
1. skillopt/engine/trainer.py — the final test-eval directory
(test_eval_final/) is written to before being created; add
os.makedirs(..., exist_ok=True), matching the two sibling test-eval dirs.
Without it, summary.json raises FileNotFoundError when a rollout yields
zero predictions.
2. skillopt/model/claude_backend.py
a. Pass the prompt via stdin (not argv): on Windows the whole command line
is capped at ~32 KB and a large optimizer prompt (the success-analyst
minibatch carrying several report trajectories) overflows it with
[WinError 206], killing the run after retries.
b. Pass the system prompt via --append-system-prompt-file (a temp file),
not argv. The system prompt here is the skill being optimized, which
SkillOpt grows over training; since the ~32 KB cap applies to the SUM of
all argv, a grown skill would re-hit [WinError 206] even with the prompt
on stdin.
c. Pin the subprocess encoding to utf-8 (errors="replace"). With text=True
and no encoding=, stdin is encoded with the system codepage; on a zh-CN
box (cp936/GBK) a prompt containing an emoji or some Latin-1 characters
raises UnicodeEncodeError before the CLI even starts, failing every retry.
3. skillopt/model/codex_backend.py — the same utf-8 encoding pin on its
subprocess.run(input=...) call (identical unpinned-encoding pattern).
4. skillopt/utils/json_utils.py — extract_json() returned None for valid-
looking JSON that strict json.loads rejects (unescaped ASCII quotes inside
CJK string values, trailing commas), silently dropping the analyst's edits
on non-schema backends (Claude/Qwen): reflect produces N edits, 0 applied.
Add a json_repair fallback, but only on a single unambiguous object — a
balanced-brace extractor plus a refuse-on-multiple-objects guard — so a
chain-of-thought "scratch + final" response can't make repair silently
return the wrong (discarded) object, which would be worse than None (None is
detectable and retryable; a wrong-but-valid edit is applied blind). Declare
json_repair in requirements.txt and the claude/qwen optional extras so the
fallback is actually present (it otherwise no-ops, dropping edits silently).
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
+2
-2
@@ -37,9 +37,9 @@ dependencies = [
|
|||||||
# Benchmark-specific dependencies
|
# Benchmark-specific dependencies
|
||||||
alfworld = ["alfworld>=0.4.0", "gymnasium>=0.29.0"]
|
alfworld = ["alfworld>=0.4.0", "gymnasium>=0.29.0"]
|
||||||
# Claude model backend
|
# Claude model backend
|
||||||
claude = ["claude-agent-sdk>=0.1.0"]
|
claude = ["claude-agent-sdk>=0.1.0", "json_repair>=0.61.0"]
|
||||||
# Qwen local model backend (via vLLM)
|
# Qwen local model backend (via vLLM)
|
||||||
qwen = ["vllm>=0.4.0"]
|
qwen = ["vllm>=0.4.0", "json_repair>=0.61.0"]
|
||||||
# SearchQA data materialization
|
# SearchQA data materialization
|
||||||
searchqa = ["datasets>=2.18.0"]
|
searchqa = ["datasets>=2.18.0"]
|
||||||
# Documentation site
|
# Documentation site
|
||||||
|
|||||||
@@ -7,6 +7,10 @@ azure-identity>=1.15.0
|
|||||||
azure-core>=1.30.0
|
azure-core>=1.30.0
|
||||||
httpx>=0.27.0
|
httpx>=0.27.0
|
||||||
|
|
||||||
|
# Tolerant JSON repair for free-form output from non-OpenAI backends
|
||||||
|
# (Claude/Qwen); without it extract_json() drops malformed analyst edits.
|
||||||
|
json_repair>=0.61.0
|
||||||
|
|
||||||
# ── Optional: ALFWorld benchmark ──────────────────
|
# ── Optional: ALFWorld benchmark ──────────────────
|
||||||
# alfworld>=0.4.0
|
# alfworld>=0.4.0
|
||||||
# gymnasium>=0.29.0
|
# gymnasium>=0.29.0
|
||||||
|
|||||||
@@ -2133,6 +2133,7 @@ class ReflACTTrainer:
|
|||||||
)
|
)
|
||||||
print(f" Test items: {test_n}")
|
print(f" Test items: {test_n}")
|
||||||
baseline_test_dir = os.path.join(out_root, "test_eval_baseline")
|
baseline_test_dir = os.path.join(out_root, "test_eval_baseline")
|
||||||
|
os.makedirs(baseline_test_dir, exist_ok=True)
|
||||||
baseline_test_results = adapter.rollout(test_env, skill_init, baseline_test_dir)
|
baseline_test_results = adapter.rollout(test_env, skill_init, baseline_test_dir)
|
||||||
baseline_test_hard, baseline_test_soft = compute_score(baseline_test_results)
|
baseline_test_hard, baseline_test_soft = compute_score(baseline_test_results)
|
||||||
baseline_buckets = _compute_task_type_buckets(baseline_test_results, task_types)
|
baseline_buckets = _compute_task_type_buckets(baseline_test_results, task_types)
|
||||||
@@ -2167,6 +2168,7 @@ class ReflACTTrainer:
|
|||||||
)
|
)
|
||||||
print(f" Test items: {test_n2}")
|
print(f" Test items: {test_n2}")
|
||||||
test_dir = os.path.join(out_root, "test_eval")
|
test_dir = os.path.join(out_root, "test_eval")
|
||||||
|
os.makedirs(test_dir, exist_ok=True)
|
||||||
test_results = adapter.rollout(test_env2, best_skill, test_dir)
|
test_results = adapter.rollout(test_env2, best_skill, test_dir)
|
||||||
test_hard, test_soft = compute_score(test_results)
|
test_hard, test_soft = compute_score(test_results)
|
||||||
best_buckets = _compute_task_type_buckets(test_results, task_types)
|
best_buckets = _compute_task_type_buckets(test_results, task_types)
|
||||||
@@ -2230,6 +2232,7 @@ class ReflACTTrainer:
|
|||||||
)
|
)
|
||||||
print(f" Test items: {test_n3}")
|
print(f" Test items: {test_n3}")
|
||||||
final_test_dir = os.path.join(out_root, "test_eval_final")
|
final_test_dir = os.path.join(out_root, "test_eval_final")
|
||||||
|
os.makedirs(final_test_dir, exist_ok=True)
|
||||||
final_test_results = adapter.rollout(test_env3, current_skill, final_test_dir)
|
final_test_results = adapter.rollout(test_env3, current_skill, final_test_dir)
|
||||||
final_test_hard, final_test_soft = compute_score(final_test_results)
|
final_test_hard, final_test_soft = compute_score(final_test_results)
|
||||||
final_buckets = _compute_task_type_buckets(final_test_results, task_types)
|
final_buckets = _compute_task_type_buckets(final_test_results, task_types)
|
||||||
|
|||||||
@@ -252,13 +252,25 @@ def _run_claude_print(*, system: str, prompt: str, model: str, tools: list[dict[
|
|||||||
if CLAUDE_SETTING_SOURCES:
|
if CLAUDE_SETTING_SOURCES:
|
||||||
cmd.extend(["--setting-sources", CLAUDE_SETTING_SOURCES])
|
cmd.extend(["--setting-sources", CLAUDE_SETTING_SOURCES])
|
||||||
if system:
|
if system:
|
||||||
cmd.extend(["--append-system-prompt", system])
|
# Write the system prompt to a file, not argv: here the skill being
|
||||||
|
# optimized IS the system prompt, and SkillOpt grows it over training,
|
||||||
|
# so past ~30 KB it would re-hit the Windows argv cap (WinError 206).
|
||||||
|
# The CLI reads it via --append-system-prompt-file.
|
||||||
|
system_path = os.path.join(temp_dir, "system_prompt.txt")
|
||||||
|
with open(system_path, "w", encoding="utf-8") as system_fh:
|
||||||
|
system_fh.write(system)
|
||||||
|
cmd.extend(["--append-system-prompt-file", system_path])
|
||||||
if effort:
|
if effort:
|
||||||
cmd.extend(["--effort", effort])
|
cmd.extend(["--effort", effort])
|
||||||
structured_output = bool(return_message)
|
structured_output = bool(return_message)
|
||||||
if structured_output:
|
if structured_output:
|
||||||
cmd.extend(["--schema", _assistant_message_schema_wrapper()])
|
cmd.extend(["--schema", _assistant_message_schema_wrapper()])
|
||||||
proc = subprocess.run(cmd + [prompt_for_cli], capture_output=True, text=True, timeout=timeout or 300, cwd=temp_dir)
|
# Feed the prompt via stdin (and the system prompt via a file, above), not
|
||||||
|
# argv: on Windows the whole command line is capped at ~32 KB and large
|
||||||
|
# optimizer prompts / grown skills overflow it → [WinError 206]. Pin UTF-8
|
||||||
|
# so a zh-CN default codepage (cp936) can't raise UnicodeEncodeError on
|
||||||
|
# emoji / non-GBK glyphs before the CLI even starts.
|
||||||
|
proc = subprocess.run(cmd, input=prompt_for_cli, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout or 300, cwd=temp_dir)
|
||||||
stderr_text = (proc.stderr or "").strip()
|
stderr_text = (proc.stderr or "").strip()
|
||||||
if proc.returncode != 0:
|
if proc.returncode != 0:
|
||||||
_check_claude_error(stderr_text, model)
|
_check_claude_error(stderr_text, model)
|
||||||
|
|||||||
@@ -328,6 +328,8 @@ def _run_codex_exec(
|
|||||||
command,
|
command,
|
||||||
input=prompt,
|
input=prompt,
|
||||||
text=True,
|
text=True,
|
||||||
|
encoding="utf-8",
|
||||||
|
errors="replace",
|
||||||
capture_output=True,
|
capture_output=True,
|
||||||
timeout=timeout,
|
timeout=timeout,
|
||||||
check=False,
|
check=False,
|
||||||
|
|||||||
@@ -3,6 +3,50 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import json
|
import json
|
||||||
import re
|
import re
|
||||||
|
import warnings
|
||||||
|
|
||||||
|
|
||||||
|
def _top_level_brace_objects(text: str) -> list[str]:
|
||||||
|
"""Return every balanced *top-level* ``{...}`` span in ``text``.
|
||||||
|
|
||||||
|
String/escape aware, so braces inside string values are not miscounted.
|
||||||
|
Used to detect ambiguity: when a response carries more than one top-level
|
||||||
|
object we must not let a repair pass silently pick one — it may pick the
|
||||||
|
wrong (discarded) edit, which is strictly worse than returning None.
|
||||||
|
"""
|
||||||
|
spans: list[str] = []
|
||||||
|
i, n = 0, len(text)
|
||||||
|
while i < n:
|
||||||
|
if text[i] != "{":
|
||||||
|
i += 1
|
||||||
|
continue
|
||||||
|
depth = 0
|
||||||
|
in_str = False
|
||||||
|
esc = False
|
||||||
|
start = i
|
||||||
|
while i < n:
|
||||||
|
ch = text[i]
|
||||||
|
if in_str:
|
||||||
|
if esc:
|
||||||
|
esc = False
|
||||||
|
elif ch == "\\":
|
||||||
|
esc = True
|
||||||
|
elif ch == '"':
|
||||||
|
in_str = False
|
||||||
|
elif ch == '"':
|
||||||
|
in_str = True
|
||||||
|
elif ch == "{":
|
||||||
|
depth += 1
|
||||||
|
elif ch == "}":
|
||||||
|
depth -= 1
|
||||||
|
if depth == 0:
|
||||||
|
spans.append(text[start:i + 1])
|
||||||
|
i += 1
|
||||||
|
break
|
||||||
|
i += 1
|
||||||
|
else:
|
||||||
|
break # unterminated final object
|
||||||
|
return spans
|
||||||
|
|
||||||
|
|
||||||
def extract_json(text: str) -> dict | None:
|
def extract_json(text: str) -> dict | None:
|
||||||
@@ -22,6 +66,39 @@ def extract_json(text: str) -> dict | None:
|
|||||||
return json.loads(m.group(0))
|
return json.loads(m.group(0))
|
||||||
except json.JSONDecodeError:
|
except json.JSONDecodeError:
|
||||||
pass
|
pass
|
||||||
|
# Tolerant fallback for non-OpenAI backends (Claude/Qwen, …) whose free-form
|
||||||
|
# JSON strict json.loads rejects — unescaped ASCII quotes inside CJK string
|
||||||
|
# values, trailing commas, etc. Repair so the analyst's edits aren't silently
|
||||||
|
# dropped, but ONLY a single unambiguous object: never feed the greedy `{.*}`
|
||||||
|
# span or the raw text, or json_repair would quietly return one of several
|
||||||
|
# objects (empirically the wrong/last one) — strictly worse than None, which
|
||||||
|
# the caller can detect and retry/skip.
|
||||||
|
try:
|
||||||
|
from json_repair import repair_json
|
||||||
|
except ModuleNotFoundError:
|
||||||
|
warnings.warn(
|
||||||
|
"json_repair not installed; malformed-JSON recovery disabled — "
|
||||||
|
"non-OpenAI analyst edits may be silently dropped. pip install json_repair",
|
||||||
|
RuntimeWarning,
|
||||||
|
stacklevel=2,
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
candidate = None
|
||||||
|
fenced = re.search(r"```json\s*(.*?)```", text, re.DOTALL)
|
||||||
|
if fenced and len(_top_level_brace_objects(fenced.group(1))) == 1:
|
||||||
|
candidate = fenced.group(1)
|
||||||
|
else:
|
||||||
|
objs = _top_level_brace_objects(text)
|
||||||
|
if len(objs) == 1:
|
||||||
|
candidate = objs[0]
|
||||||
|
# 0 or >1 top-level objects → too ambiguous to repair safely → None
|
||||||
|
if candidate:
|
||||||
|
try:
|
||||||
|
repaired = repair_json(candidate, return_objects=True)
|
||||||
|
if isinstance(repaired, dict) and repaired:
|
||||||
|
return repaired
|
||||||
|
except Exception: # noqa: BLE001 — repair is best-effort
|
||||||
|
pass
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user