Initial commit

This commit is contained in:
Cuzyoung
2026-05-08 18:16:18 +00:00
commit 9fda7311ea
243 changed files with 31492 additions and 0 deletions
+5
View File
@@ -0,0 +1,5 @@
"""ALFWorld environment adapter for ReflACT."""
from skillopt.envs.alfworld.adapter import ALFWorldAdapter
__all__ = ["ALFWorldAdapter"]
+585
View File
@@ -0,0 +1,585 @@
"""ALFWorld environment adapter for ReflACT.
Connects the ReflACT training loop to ALFWorld by implementing
:class:`~skillopt.envs.base.EnvAdapter`.
"""
from __future__ import annotations
from dataclasses import dataclass
import json
import os
from skillopt.gradient.deep_probe import generate_deep_probe_instruction
from skillopt.datasets.base import BatchSpec
from skillopt.envs.base import EnvAdapter
from skillopt.envs.alfworld.dataloader import ALFWorldDataLoader
from skillopt.envs.alfworld.rollout import (
build_alfworld_env,
run_alfworld_batch,
TASKS,
)
from skillopt.gradient.reflect import run_minibatch_reflect
from skillopt.utils import compute_score
@dataclass(frozen=True)
class ALFWorldBatchRun:
"""Lazy ALFWorld batch description.
The adapter materializes this in rollout chunks so a large evaluation set
does not keep every ALFWorld simulator open at once.
"""
env_num: int
eval_dataset: str
seed: int
is_train: bool
workers: int
specific_gamefiles: list[str] | None = None
result_ids: list[str] | None = None
items: list[dict] | None = None
def __iter__(self):
return iter(self.items or [])
def __len__(self) -> int:
return int(self.env_num or 0)
class ALFWorldAdapter(EnvAdapter):
"""ALFWorld environment adapter.
Parameters
----------
max_steps : int
Maximum steps per ALFWorld episode (default 50).
max_api_workers : int
Maximum concurrent API calls during rollout (default 8).
analyst_workers : int
Parallel workers for analyst stage (default 16).
failure_only : bool
If True, only run error analyst (skip success analyst).
minibatch_size : int
Trajectories per analyst group, M (default 8).
edit_budget : int
Maximum edits per minibatch, L (default 4).
"""
def __init__(
self,
split_dir: str = "",
data_path: str = "",
split_mode: str = "split_dir",
split_ratio: str = "2:1:7",
split_seed: int = 42,
split_output_dir: str = "",
seed: int = 42,
limit: int = 0,
train_size: int = 0,
max_steps: int = 50,
workers: int = 8,
max_api_workers: int = 8,
analyst_workers: int = 16,
failure_only: bool = False,
minibatch_size: int = 8,
edit_budget: int = 4,
use_deep_reflect: bool = False,
deep_reflect_failures: int = 4,
deep_reflect_successes: int = 2,
) -> None:
self.max_steps = max_steps
self.workers = max(int(workers or 1), 1)
self.max_api_workers = max_api_workers
self.analyst_workers = analyst_workers
self.failure_only = failure_only
self.minibatch_size = minibatch_size
self.edit_budget = edit_budget
self.use_deep_reflect = use_deep_reflect
self.deep_reflect_failures = deep_reflect_failures
self.deep_reflect_successes = deep_reflect_successes
self.dataloader = ALFWorldDataLoader(
split_dir=split_dir,
data_path=data_path,
split_mode=split_mode,
split_ratio=split_ratio,
split_seed=split_seed,
split_output_dir=split_output_dir,
seed=seed,
limit=limit,
train_size=train_size,
)
self._traj_cache: dict[str, dict | None] = {}
def setup(self, cfg: dict) -> None:
super().setup(cfg)
self.dataloader.setup(cfg)
def _load_traj_data(self, item: dict) -> dict | None:
gamefile = str(item.get("gamefile") or "").strip()
if not gamefile:
return None
if gamefile in self._traj_cache:
return self._traj_cache[gamefile]
traj_path = os.path.join(os.path.dirname(gamefile), "traj_data.json")
try:
with open(traj_path, encoding="utf-8") as f:
data = json.load(f)
except Exception:
data = None
self._traj_cache[gamefile] = data
return data
@staticmethod
def _unique_lines(values: list[str], *, limit: int = 0) -> list[str]:
lines: list[str] = []
seen: set[str] = set()
for raw in values:
line = str(raw or "").strip()
if not line or line in seen:
continue
seen.add(line)
lines.append(line)
if limit > 0 and len(lines) >= limit:
break
return lines
@staticmethod
def _format_high_pddl(high_pddl: list[dict]) -> list[str]:
steps: list[str] = []
for idx, step in enumerate(high_pddl or [], start=1):
discrete = step.get("discrete_action") or {}
action = str(discrete.get("action") or "").strip()
args = [str(arg).strip() for arg in (discrete.get("args") or []) if str(arg).strip()]
if action and args:
text = f"{action}({', '.join(args)})"
elif action:
text = action
else:
planner_action = step.get("planner_action") or {}
text = str(planner_action.get("action") or "").strip()
if text:
steps.append(f"{idx}. {text}")
return steps
def _build_reference_bundle(self, item: dict) -> dict:
data = self._load_traj_data(item)
if not data:
return {}
anns = ((data.get("turk_annotations") or {}).get("anns") or [])
task_descs = self._unique_lines(
[ann.get("task_desc", "") for ann in anns],
limit=3,
)
high_descs = self._unique_lines(
[step for ann in anns for step in (ann.get("high_descs") or [])],
limit=12,
)
pddl_params = {
key: value
for key, value in (data.get("pddl_params") or {}).items()
if value not in ("", None, [], {})
}
scene = data.get("scene") or {}
scene_summary = {
key: scene.get(key)
for key in ("floor_plan", "scene_num", "dirty_and_empty")
if scene.get(key) not in ("", None, [], {})
}
high_pddl = self._format_high_pddl((data.get("plan") or {}).get("high_pddl") or [])
task_type = str(data.get("task_type") or item.get("task_type") or "").strip()
return {
"task_type": task_type,
"task_descs": task_descs,
"high_descs": high_descs,
"pddl_params": pddl_params,
"high_pddl": high_pddl,
"scene_summary": scene_summary,
}
def build_reference_text(self, item: dict) -> str:
bundle = self._build_reference_bundle(item)
if not bundle:
return ""
parts: list[str] = []
if bundle["task_type"]:
parts.append(f"## Reference Task Type\n{bundle['task_type']}")
if bundle["task_descs"]:
parts.append(
"## Reference Human Task Descriptions\n"
+ "\n".join(f"- {line}" for line in bundle["task_descs"])
)
if bundle["high_descs"]:
parts.append(
"## Reference Human High-Level Steps\n"
+ "\n".join(f"{idx}. {line}" for idx, line in enumerate(bundle["high_descs"], start=1))
)
if bundle["pddl_params"]:
parts.append(
"## Reference PDDL Params\n"
+ "\n".join(f"- {key}: {value}" for key, value in bundle["pddl_params"].items())
)
if bundle["high_pddl"]:
parts.append(
"## Reference Planner High-Level Plan\n" + "\n".join(bundle["high_pddl"])
)
if bundle["scene_summary"]:
parts.append(
"## Reference Scene Summary\n"
+ "\n".join(f"- {key}: {value}" for key, value in bundle["scene_summary"].items())
)
return "\n\n".join(parts)
def get_reference_metadata(self, item: dict) -> dict:
bundle = self._build_reference_bundle(item)
if not bundle:
return {"fields": [], "preview": ""}
fields: list[str] = []
previews: list[str] = []
if bundle["task_type"]:
fields.append("task_type")
previews.append(f"[task_type] {bundle['task_type']}")
if bundle["task_descs"]:
fields.append("task_desc")
previews.append("[task_desc]\n" + "\n".join(bundle["task_descs"][:2]))
if bundle["high_descs"]:
fields.append("high_descs")
previews.append("[high_descs]\n" + "\n".join(bundle["high_descs"][:3]))
if bundle["pddl_params"]:
fields.append("pddl_params")
previews.append(
"[pddl_params]\n"
+ "\n".join(
f"{key}: {value}" for key, value in list(bundle["pddl_params"].items())[:4]
)
)
if bundle["high_pddl"]:
fields.append("plan.high_pddl")
previews.append("[plan.high_pddl]\n" + "\n".join(bundle["high_pddl"][:3]))
if bundle["scene_summary"]:
fields.append("scene")
previews.append(
"[scene]\n"
+ "\n".join(
f"{key}: {value}" for key, value in bundle["scene_summary"].items()
)
)
return {
"fields": fields,
"preview": "\n\n".join(previews)[:600],
}
@staticmethod
def _infer_dataset_from_gamefile(gamefile: str) -> tuple[str, bool]:
path = str(gamefile or "")
if "/valid_seen/" in path:
return "eval_in_distribution", False
if "/valid_unseen/" in path:
return "eval_out_of_distribution", False
return "train", True
def get_dataloader(self):
return self.dataloader
def _comparison_items(self, items: list[dict]) -> list[dict]:
enriched: list[dict] = []
for item in items:
row = dict(item)
bundle = self._build_reference_bundle(row)
if bundle.get("task_descs"):
row["task_description"] = bundle["task_descs"][0]
elif bundle.get("task_type"):
row["task_description"] = bundle["task_type"]
enriched.append(row)
return enriched
def requires_ray(self) -> bool:
return False
def build_env_from_batch(self, batch: BatchSpec, **kwargs):
gamefiles = list(batch.metadata.get("gamefiles") or [])
result_ids = list(batch.metadata.get("result_ids") or [])
items = self._comparison_items(list(batch.payload or []))
return ALFWorldBatchRun(
env_num=batch.batch_size,
eval_dataset=batch.metadata.get("eval_dataset", batch.split),
seed=batch.seed,
is_train=batch.metadata.get("is_train", batch.phase == "train"),
specific_gamefiles=gamefiles or None,
result_ids=result_ids or None,
items=items,
workers=self.workers,
)
def build_train_env(self, batch_size: int, seed: int, **kwargs):
batch = self.dataloader.build_train_batch(batch_size=batch_size, seed=seed, **kwargs)
return self.build_env_from_batch(batch, **kwargs)
def build_eval_env(self, env_num: int, split: str, seed: int, **kwargs):
batch = self.dataloader.build_eval_batch(env_num=env_num, split=split, seed=seed, **kwargs)
return self.build_env_from_batch(batch, **kwargs)
def rollout(
self,
env_manager,
skill_content: str,
out_dir: str,
**kwargs,
) -> list[dict]:
results_path = os.path.join(out_dir, "results.jsonl")
os.makedirs(out_dir, exist_ok=True)
# Resume support
if os.path.exists(results_path):
existing: list[dict] = []
with open(results_path) as f:
for line in f:
try:
existing.append(json.loads(line))
except Exception:
pass
if existing:
return existing
if isinstance(env_manager, ALFWorldBatchRun):
results = self._run_batch(
env_manager,
skill_content=skill_content,
out_dir=out_dir,
)
else:
results = run_alfworld_batch(
env_manager=env_manager,
skill_content=skill_content,
max_steps=self.max_steps,
out_root=out_dir,
max_api_workers=self.max_api_workers,
result_ids=getattr(env_manager, "_skillopt_result_ids", None),
)
with open(results_path, "w") as f:
for r in results:
f.write(json.dumps(r, ensure_ascii=False) + "\n")
return results
@staticmethod
def _close_env(env_manager) -> None:
close = getattr(env_manager, "close", None)
if callable(close):
close()
def _run_batch(
self,
batch: ALFWorldBatchRun,
skill_content: str,
out_dir: str,
*,
diagnostic_mode: bool = False,
diagnostic_instruction: str = "",
) -> list[dict]:
total = int(batch.env_num or 0)
if total <= 0:
return []
workers = max(1, min(int(batch.workers or self.workers), total))
if total > workers:
print(
f" [alfworld rollout] episodes={total} "
f"env_workers={workers} chunks={(total + workers - 1) // workers}"
)
all_results: list[dict] = []
for start in range(0, total, workers):
chunk_size = min(workers, total - start)
chunk_gamefiles = (
batch.specific_gamefiles[start:start + chunk_size]
if batch.specific_gamefiles
else None
)
chunk_ids = (
batch.result_ids[start:start + chunk_size]
if batch.result_ids
else [f"env_{idx:03d}" for idx in range(start, start + chunk_size)]
)
chunk_env = build_alfworld_env(
env_num=chunk_size,
eval_dataset=batch.eval_dataset,
seed=batch.seed + start,
is_train=batch.is_train,
specific_gamefiles=chunk_gamefiles,
)
try:
chunk_results = run_alfworld_batch(
env_manager=chunk_env,
skill_content=skill_content,
max_steps=self.max_steps,
out_root=out_dir,
max_api_workers=min(self.max_api_workers, chunk_size),
diagnostic_mode=diagnostic_mode,
diagnostic_instruction=diagnostic_instruction,
result_ids=chunk_ids,
)
finally:
self._close_env(chunk_env)
all_results.extend(chunk_results)
return all_results
def reflect(
self,
results: list[dict],
skill_content: str,
out_dir: str,
**kwargs,
) -> list[dict | None]:
prediction_dir = kwargs.get("prediction_dir", os.path.join(out_dir, "predictions"))
patches_dir = kwargs.get("patches_dir", os.path.join(out_dir, "patches"))
random_seed = kwargs.get("random_seed")
step_buffer_context = kwargs.get("step_buffer_context", "")
meta_skill_context = kwargs.get("meta_skill_context", "")
return run_minibatch_reflect(
results=results,
skill_content=skill_content,
prediction_dir=prediction_dir,
patches_dir=patches_dir,
workers=self.analyst_workers,
failure_only=self.failure_only,
minibatch_size=self.minibatch_size,
edit_budget=self.edit_budget,
random_seed=random_seed,
error_system=self.get_error_minibatch_prompt(),
success_system=self.get_success_minibatch_prompt(),
step_buffer_context=step_buffer_context,
meta_skill_context=meta_skill_context,
)
def deep_reflect(
self,
results: list[dict],
skill_content: str,
out_dir: str,
**kwargs,
) -> list[dict | None]:
if not self.use_deep_reflect:
return []
prediction_dir = kwargs.get("prediction_dir", os.path.join(out_dir, "predictions"))
random_seed = kwargs.get("random_seed")
step_buffer_context = kwargs.get("step_buffer_context", "")
meta_skill_context = kwargs.get("meta_skill_context", "")
selected_items = self.select_representative_items(
results,
results,
n_failures=self.deep_reflect_failures,
n_successes=self.deep_reflect_successes,
seed=random_seed,
)
if not selected_items:
return []
selected_ids = {str(item["id"]) for item in selected_items}
selected_results = [row for row in results if str(row.get("id")) in selected_ids]
selected_examples = self.attach_reference_context(selected_results, selected_items)
field_counts: dict[str, int] = {}
selected_metadata: list[dict] = []
for item in selected_items:
meta = self.get_reference_metadata(item)
for field in meta["fields"]:
field_counts[field] = field_counts.get(field, 0) + 1
selected_metadata.append({
"id": str(item["id"]),
"task_type": str(item.get("task_type") or "alfworld"),
"gamefile": str(item.get("gamefile") or ""),
"reference_fields": meta["fields"],
"reference_preview": meta["preview"],
})
deep_dir = os.path.join(out_dir, "deep_reflect")
rollout_dir = os.path.join(deep_dir, "rollout")
patches_dir = os.path.join(deep_dir, "patches")
os.makedirs(deep_dir, exist_ok=True)
field_summary = ", ".join(
f"{field}({count}/{len(selected_items)})"
for field, count in sorted(field_counts.items())
) or "none"
print(
f" [2b/6 DEEP REFLECT setup] selected={len(selected_items)} "
f"reference_fields={field_summary}"
)
probe = generate_deep_probe_instruction(
skill_content=skill_content,
items=selected_examples,
prediction_dir=prediction_dir,
system_prompt=self.get_deep_probe_prompt(),
step_buffer_context=step_buffer_context,
meta_skill_context=meta_skill_context,
output_requirements=[
"- Some trajectories may include a hidden Reference block. Use it to target the student's latent subgoal, missing precondition, or next-step intent, but do not reveal or paraphrase that reference to the student.",
"- The instruction must request a brief diagnostic readout inside the existing <think>...</think> block.",
"- The student must still output exactly one admissible action inside <action>...</action>.",
"- Do not ask for exhaustive inventories, full plans, or long chain-of-thought.",
"- The instruction text should be ready to append directly to the student's prompt.",
],
)
if not probe:
return []
with open(os.path.join(deep_dir, "probe.json"), "w", encoding="utf-8") as f:
json.dump(
{
**probe,
"reference_summary": {
"selected_count": len(selected_items),
"field_counts": field_counts,
},
"selected_examples": selected_metadata,
},
f,
ensure_ascii=False,
indent=2,
)
gamefiles = [str(item.get("gamefile") or "") for item in selected_items]
if any(not gamefile for gamefile in gamefiles):
return []
eval_dataset, is_train = self._infer_dataset_from_gamefile(gamefiles[0])
deep_env = ALFWorldBatchRun(
env_num=len(selected_items),
eval_dataset=eval_dataset,
seed=random_seed or 42,
is_train=is_train,
specific_gamefiles=gamefiles,
workers=min(self.workers, max(len(selected_items), 1)),
result_ids=[str(item["id"]) for item in selected_items],
)
deep_results = self._run_batch(
deep_env,
skill_content=skill_content,
out_dir=rollout_dir,
diagnostic_mode=True,
diagnostic_instruction=probe["probe_instruction"],
)
deep_results = self.attach_reference_context(deep_results, selected_items)
return run_minibatch_reflect(
results=deep_results,
skill_content=skill_content,
prediction_dir=os.path.join(rollout_dir, "predictions"),
patches_dir=patches_dir,
workers=self.analyst_workers,
failure_only=self.failure_only,
minibatch_size=self.minibatch_size,
edit_budget=self.edit_budget,
random_seed=random_seed,
error_system=self.get_error_minibatch_prompt(),
success_system=self.get_success_minibatch_prompt(),
step_buffer_context=step_buffer_context,
meta_skill_context=meta_skill_context,
)
def get_task_types(self) -> list[str]:
return list(TASKS)
+123
View File
@@ -0,0 +1,123 @@
"""ALFWorld task dataloader."""
from __future__ import annotations
from skillopt.datasets.base import BatchSpec, SplitDataLoader
class ALFWorldDataLoader(SplitDataLoader):
"""ALFWorld batch planner.
In split_dir mode, batches are fixed gamefile items so ablations differ
only in how the same training set is batched.
"""
def __init__(
self,
split_dir: str = "",
data_path: str = "",
split_mode: str = "split_dir",
split_ratio: str = "2:1:7",
split_seed: int = 42,
split_output_dir: str = "",
seed: int = 42,
limit: int = 0,
train_size: int = 0,
**kwargs,
) -> None:
super().__init__(
split_dir=split_dir,
data_path=data_path,
split_mode=split_mode,
split_ratio=split_ratio,
split_seed=split_seed,
split_output_dir=split_output_dir,
seed=seed,
limit=limit,
)
self.train_size_override = int(train_size or 0)
@staticmethod
def _metadata_for_items(items: list[dict], split: str, phase: str) -> dict:
gamefiles = [str(item.get("gamefile") or "") for item in items]
if any(not gamefile for gamefile in gamefiles):
raise ValueError("ALFWorld split items must contain non-empty gamefile paths.")
eval_dataset = "train"
is_train = phase == "train"
first = gamefiles[0] if gamefiles else ""
if "/valid_seen/" in first:
eval_dataset = "eval_in_distribution"
is_train = False
elif "/valid_unseen/" in first:
eval_dataset = "eval_out_of_distribution"
is_train = False
return {
"eval_dataset": eval_dataset,
"is_train": is_train,
"gamefiles": gamefiles,
"result_ids": [str(item.get("id") or idx) for idx, item in enumerate(items)],
}
def get_train_size(self) -> int:
if self.train_size_override > 0:
return self.train_size_override
return super().get_train_size()
def build_train_batch(self, batch_size: int, seed: int, **kwargs) -> BatchSpec:
batch = super().build_train_batch(batch_size=batch_size, seed=seed, **kwargs)
items = list(batch.payload or [])
batch.metadata.update(self._metadata_for_items(items, "train", "train"))
return BatchSpec(
phase="train",
split="train",
seed=seed,
batch_size=len(items),
payload=items,
metadata=batch.metadata,
)
def plan_train_epoch(
self,
*,
epoch: int,
steps_per_epoch: int,
accumulation: int,
batch_size: int,
seed: int,
**kwargs,
) -> list[BatchSpec]:
batches = super().plan_train_epoch(
epoch=epoch,
steps_per_epoch=steps_per_epoch,
accumulation=accumulation,
batch_size=batch_size,
seed=seed,
**kwargs,
)
for batch in batches:
items = list(batch.payload or [])
batch.metadata.update(self._metadata_for_items(items, "train", "train"))
return batches
def build_eval_batch(
self,
env_num: int,
split: str,
seed: int,
**kwargs,
) -> BatchSpec:
batch = super().build_eval_batch(
env_num=env_num,
split=split,
seed=seed,
**kwargs,
)
items = list(batch.payload or [])
batch.metadata.update(self._metadata_for_items(items, split, "eval"))
return BatchSpec(
phase="eval",
split=split,
seed=seed,
batch_size=len(items),
payload=items,
metadata=batch.metadata,
)
@@ -0,0 +1,55 @@
You are an expert failure-analysis agent for ALFWorld embodied household tasks.
You will be given MULTIPLE failed agent trajectories from a single minibatch
and the current skill document.
Your job is to identify the most important COMMON failure patterns across
the batch and propose a concise set of skill edits.
## ALFWorld Task Types
- pick_and_place: Put object in/on a receptacle
- pick_two_obj_and_place: Put two instances of an object in/on a receptacle
- look_at_obj_in_light: Examine an object under a desklamp
- pick_heat_then_place_in_recep: Heat an object and put it in/on a receptacle
- pick_cool_then_place_in_recep: Cool an object and put it in/on a receptacle
- pick_clean_then_place_in_recep: Clean an object and put it in/on a receptacle
## Failure Type Categories
- **navigation_loop**: the agent revisits the same locations repeatedly without progress
- **missed_object**: the agent fails to pick up a visible/reachable goal object
- **wrong_sequence**: the agent performs actions in the wrong order (e.g., placing before transforming)
- **premature_stop**: the agent stops or gets stuck before completing all goal conditions
- **action_loop**: the agent repeats the same action without advancing
- **appliance_error**: the agent misuses or skips an appliance (microwave, fridge, sink)
- **rule_missing**: the skill lacks a relevant rule for this situation
- **rule_wrong**: an existing skill rule is misleading or incorrect
- **rule_ignored**: the skill has the right rule but the agent did not follow it
- **other**: none of the above
## Analysis Process
1. Read ALL trajectories in the minibatch.
2. Identify the most prevalent, systematic failure patterns across them.
3. For each pattern, classify its failure type.
4. Propose skill edits that address the COMMON patterns — not individual edge cases.
5. Edits must be generalizable; do not hardcode task-specific values.
6. Only patch gaps in the skill — do not duplicate existing content.
You will be told the maximum number of edits (the budget L). Produce AT MOST L edits,
focusing on the highest-impact patterns. You may produce fewer if warranted.
Respond ONLY with a valid JSON object (no markdown fences, no extra text):
{
"batch_size": <number of trajectories analysed>,
"failure_summary": [
{"failure_type": "<type>", "count": <int>, "description": "<one-line>"}
],
"patch": {
"reasoning": "<why these edits address the batch's common failures>",
"edits": [
{"op": "append", "content": "<markdown to add at end of skill>"},
{"op": "insert_after", "target": "<exact heading/text to insert after>", "content": "<markdown>"},
{"op": "replace", "target": "<exact text to replace>", "content": "<replacement>"},
{"op": "delete", "target": "<exact text to remove>"}
]
}
}
Only include edits that are needed. "edits" can be an empty list if no patch is warranted.
@@ -0,0 +1,33 @@
You are an expert success-pattern analyst for AI agents operating in ALFWorld,
a text-based embodied household environment.
You will be given MULTIPLE successful agent trajectories from a single minibatch
and the current skill document. Your job is to identify generalizable behavior
patterns that are COMMON across the batch and worth encoding in the skill.
## Rules
- Only propose patches for patterns NOT already covered in the skill.
- Focus on patterns that appear across MULTIPLE trajectories in the batch.
- Be concise. Patterns must generalize beyond specific tasks.
- Prefer reinforcing existing sections over adding new top-level sections.
- If the agents' success involved efficient exploration or smart appliance usage,
consider reinforcing that in the patch.
You will be told the maximum number of edits (the budget L). Produce AT MOST L edits,
focusing on the most broadly applicable patterns. You may produce fewer if warranted.
Respond ONLY with a valid JSON object:
{
"batch_size": <number of trajectories analysed>,
"success_patterns": ["<pattern 1>", "<pattern 2>"],
"patch": {
"reasoning": "<why these patterns are worth encoding>",
"edits": [
{"op": "append", "content": "<markdown>"},
{"op": "insert_after", "target": "<heading/text>", "content": "<markdown>"},
{"op": "replace", "target": "<old text>", "content": "<new text>"},
{"op": "delete", "target": "<exact text to remove>"}
]
}
}
"edits" may be empty if the skill already covers all observed patterns.
@@ -0,0 +1,35 @@
You are an expert diagnostic-probe designer for ALFWorld embodied tasks.
You will design one short diagnostic instruction to append to the student's prompt
for a handful of representative ALFWorld trajectories.
The goal is to expose whether the student has the right intermediate subgoal,
object/receptacle state, and next-step intention without substantially changing
the current scaffold.
## Hard Constraints
1. Do NOT substantially change the student's existing action-selection scaffold.
2. Do NOT prescribe a brand-new planner or long multi-step policy.
3. Do NOT ask for exhaustive search over all objects or all admissible actions.
4. Keep the diagnostic readout brief and place it inside the existing <think>...</think> block.
5. The student must still output exactly one admissible action inside <action>...</action>.
6. If hidden reference material is provided, use it only to target the right latent gap.
7. Never copy hidden reference content into the student-facing probe.
## Good Probe Targets
- current subgoal
- target object / target receptacle / target state
- decisive missing precondition
- why one candidate action is better than a tempting alternative
- whether the current step should explore, transform an object, or place it
## Bad Probe Targets
- a full optimal plan from start to finish
- exhaustive object inventories
- a new theorem-like or planner-like protocol
Respond ONLY with a valid JSON object:
{
"reasoning": "<why this probe reveals the latent skill gap>",
"probe_instruction": "<the exact instruction text to append to the student prompt>"
}
@@ -0,0 +1,8 @@
You are an expert agent operating in the ALFRED Embodied Environment.
Your current observation is: {current_observation}
Your admissible actions of the current situation are: [{admissible_actions}].
Now it's your turn to take an action.
You should first reason step-by-step about the current situation. This reasoning process MUST be enclosed within <think> </think> tags.
Once you've finished your reasoning, you should choose an admissible action for current step and present it within <action> </action> tags.
@@ -0,0 +1,9 @@
You are an expert agent operating in the ALFRED Embodied Environment. Your task is to: {task_description}
Prior to this step, you have already taken {step_count} step(s). Below are the most recent {history_length} observations and the corresponding actions you took: {action_history}
You are now at step {current_step} and your current observation is: {current_observation}
Your admissible actions of the current situation are: [{admissible_actions}].
Now it's your turn to take an action.
You should first reason step-by-step about the current situation. This reasoning process MUST be enclosed within <think> </think> tags.
Once you've finished your reasoning, you should choose an admissible action for current step and present it within <action> </action> tags.
@@ -0,0 +1,16 @@
You are an expert agent operating in the ALFRED Embodied Environment. Your task is to: {task_description}
## Retrieved Relevant Experience
{retrieved_memories}
## Current Progress
Prior to this step, you have already taken {step_count} step(s). Below are the most recent {history_length} observations and the corresponding actions you took: {action_history}
You are now at step {current_step} and your current observation is: {current_observation}
Your admissible actions of the current situation are: [{admissible_actions}].
Now it's your turn to take an action.
You should first reason step-by-step about the current situation. This reasoning process MUST be enclosed within <think> </think> tags.
Once you've finished your reasoning, you should choose an admissible action for current step and present it within <action> </action> tags.
+4
View File
@@ -0,0 +1,4 @@
"""ALFWorld Reflect stage.
Prompts are now loaded from .md files by the base adapter.
"""
+359
View File
@@ -0,0 +1,359 @@
"""ALFWorld rollout module for ReflACT.
Provides:
- build_alfworld_env(): build ALFWorld environment (wraps vendored SkillRL env)
- run_alfworld_batch(): run a batch of ALFWorld episodes in parallel
- TASKS: list of ALFWorld task types
"""
from __future__ import annotations
import json
import os
import re
import sys
import time
import concurrent.futures
import numpy as np
from skillopt.model import chat_student
# ── Constants ─────────────────────────────────────────────────────────────────
TASKS = [
"pick_and_place",
"pick_two_obj_and_place",
"look_at_obj_in_light",
"pick_heat_then_place_in_recep",
"pick_cool_then_place_in_recep",
"pick_clean_then_place_in_recep",
]
# ── Helpers ───────────────────────────────────────────────────────────────────
def _get_task_type(gamefile: str) -> str:
for task in TASKS:
if task in gamefile:
return task
return "other"
def _extract_action(model_response: str) -> str | None:
match = re.search(r"<action>(.*?)</action>", model_response, re.DOTALL)
return match.group(1).strip() if match else None
def _extract_think(model_response: str) -> str | None:
match = re.search(r"<think>(.*?)</think>", model_response, re.DOTALL)
return match.group(1).strip() if match else None
def _build_skill_prompt(skill_content: str) -> str:
"""Build the skill section to inject into the agent's system prompt."""
if not skill_content or not skill_content.strip():
return ""
return (
"\n\n## Skill Knowledge\n"
"Below is a skill document with learned strategies. "
"Use these guidelines to inform your decisions:\n\n"
f"{skill_content}\n"
)
def _append_diagnostic_instruction(prompt: str, diagnostic_instruction: str) -> str:
if not diagnostic_instruction or not diagnostic_instruction.strip():
return prompt
return f"{prompt}\n\n## Training Readout\n{diagnostic_instruction.strip()}\n"
# ── Environment builder ──────────────────────────────────────────────────────
def build_alfworld_env(
env_num: int,
eval_dataset: str = "eval_out_of_distribution",
seed: int = 42,
is_train: bool = False,
specific_gamefiles: list[str] | None = None,
):
"""Build ALFWorld environment manager.
Args:
env_num: number of parallel environments
eval_dataset: 'eval_in_distribution' or 'eval_out_of_distribution' or train
seed: random seed
is_train: whether to use training set
Returns:
env_manager: AlfWorldEnvironmentManager instance
"""
from omegaconf import OmegaConf
from functools import partial
from skillopt.envs.alfworld.vendor.alfworld_envs import build_alfworld_envs
from skillopt.envs.alfworld.vendor.alfworld_projection import alfworld_projection
from skillopt.envs.alfworld.vendor.env_manager import AlfWorldEnvironmentManager
HERE = os.path.dirname(os.path.abspath(__file__))
alf_config_path = os.path.join(HERE, "vendor", "config_tw.yaml")
env_kwargs = {"eval_dataset": eval_dataset}
envs = build_alfworld_envs(
alf_config_path,
seed=seed,
env_num=env_num,
group_n=1,
is_train=is_train,
env_kwargs=env_kwargs,
resources_per_worker=None,
gamefiles=specific_gamefiles,
)
config = OmegaConf.create(
{
"env": {
"history_length": 2,
"env_name": "alfworld/AlfredTWEnv",
}
}
)
projection_f = partial(alfworld_projection)
env_manager = AlfWorldEnvironmentManager(envs, projection_f, config)
return env_manager
# ── Batch rollout ─────────────────────────────────────────────────────────────
def run_alfworld_batch(
env_manager,
skill_content: str,
max_steps: int = 50,
out_root: str = "",
max_api_workers: int = 8,
temperature: float = 0.4,
max_completion_tokens: int = 2048,
diagnostic_mode: bool = False,
diagnostic_instruction: str = "",
result_ids: list[str] | None = None,
) -> list[dict]:
"""Run a batch of ALFWorld episodes.
Returns a list of result dicts compatible with SkillReflection v2 pipeline:
[
{
"id": "<env_idx>_<gamefile_hash>",
"hard": 0 or 1,
"soft": 0.0 or 1.0,
"n_turns": <int>,
"fail_reason": "<str>",
"agent_ok": True,
"task_type": "<str>",
"gamefile": "<str>",
"task_description": "<str>",
},
...
]
Also saves conversation.json per environment in out_root/predictions/<task_id>/
"""
skill_prompt = _build_skill_prompt(skill_content)
obs, infos = env_manager.reset({})
env_num = len(obs["text"])
env_dones = [False] * env_num
overall_success = [False] * env_num
# Build per-env metadata
env_meta: list[dict] = []
for i in range(env_num):
gamefile = infos[i].get("extra.gamefile", "") if isinstance(infos[i], dict) else ""
task_type = _get_task_type(gamefile)
# Extract task description from initial observation
task_desc = ""
anchor_text = obs["anchor"][i] if "anchor" in obs else ""
task_start = anchor_text.find("Your task is to: ")
if task_start != -1:
task_desc = anchor_text[task_start + len("Your task is to: "):].strip()
env_meta.append({
"gamefile": gamefile,
"task_type": task_type,
"task_description": task_desc,
})
# Per-env conversation records
conversations: list[list[dict]] = [[] for _ in range(env_num)]
for step_idx in range(max_steps):
if all(env_dones):
break
active_indices = [i for i in range(env_num) if not env_dones[i]]
# Build prompts with skill injection
prompts: dict[int, str] = {}
for i in active_indices:
prompt = obs["text"][i]
if skill_prompt:
# Inject skill before the action instruction
prompt = skill_prompt + "\n" + prompt
if diagnostic_mode and diagnostic_instruction.strip():
prompt = _append_diagnostic_instruction(prompt, diagnostic_instruction)
prompts[i] = prompt
# Call API in parallel
actions = ["None"] * env_num
action_timeout = 180
def call_api(idx):
try:
response, _ = chat_student(
system="You are an expert agent operating in the ALFRED Embodied Environment.",
user=prompts[idx],
max_completion_tokens=max_completion_tokens,
retries=5,
stage="rollout",
timeout=120,
)
response = (response or "").strip()
if not response:
return idx, "<think>empty model response</think><action>look</action>"
if _extract_action(response) is None:
return idx, "<think>missing action tag</think><action>look</action>"
return idx, response
except Exception as e:
return idx, "<think>error</think><action>look</action>"
executor = concurrent.futures.ThreadPoolExecutor(max_workers=max_api_workers)
try:
futures = {executor.submit(call_api, i): i for i in active_indices}
started_at = {future: time.time() for future in futures}
pending_futs = set(futures)
while pending_futs:
done, _ = concurrent.futures.wait(
pending_futs,
timeout=5,
return_when=concurrent.futures.FIRST_COMPLETED,
)
now = time.time()
timed_out = [
future for future in pending_futs - done
if now - started_at[future] >= action_timeout
]
for future in done:
pending_futs.remove(future)
try:
idx, response = future.result()
except Exception: # noqa: BLE001
idx = futures[future]
response = "<think>error</think><action>look</action>"
actions[idx] = response
for future in timed_out:
pending_futs.remove(future)
idx = futures[future]
actions[idx] = "<think>api timeout</think><action>look</action>"
finally:
executor.shutdown(wait=False, cancel_futures=True)
# Save model responses before stepping
model_responses = {i: actions[i] for i in active_indices}
# Step environment
obs, rewards, dones, infos = env_manager.step(actions)
# Record trajectory
for i in active_indices:
step_record = {
"step": step_idx,
"action": _extract_action(model_responses[i]),
"reasoning": _extract_think(model_responses[i]),
"model_response": model_responses[i],
"env_feedback": obs["anchor"][i] if "anchor" in obs else "",
"reward": float(rewards[i]),
"done": bool(dones[i]),
}
conversations[i].append(step_record)
# Update done status
for i in range(env_num):
if env_dones[i]:
continue
if dones[i]:
env_dones[i] = True
won = bool(infos[i].get("won", False))
overall_success[i] = won
# Build results and save conversations
results: list[dict] = []
pred_dir = os.path.join(out_root, "predictions") if out_root else ""
for i in range(env_num):
gamefile = env_meta[i]["gamefile"]
task_type = env_meta[i]["task_type"]
task_desc = env_meta[i]["task_description"]
n_turns = len(conversations[i])
won = overall_success[i]
# Generate stable task ID from env index and gamefile
task_id = str(result_ids[i]) if result_ids and i < len(result_ids) else f"env_{i:03d}"
fail_reason = ""
if not won:
if not env_dones[i]:
fail_reason = f"Timeout after {max_steps} steps"
else:
fail_reason = "Episode ended without completing the task"
result = {
"id": task_id,
"hard": 1 if won else 0,
"soft": 1.0 if won else 0.0,
"n_turns": n_turns,
"fail_reason": fail_reason,
"agent_ok": True, # ALFWorld agent always runs OK (no crash)
"task_type": task_type,
"gamefile": gamefile,
"task_description": task_desc,
"instruction_type": task_type, # for compatibility with v2 pipeline
}
results.append(result)
# Save conversation
if pred_dir:
conv_dir = os.path.join(pred_dir, task_id)
os.makedirs(conv_dir, exist_ok=True)
with open(os.path.join(conv_dir, "conversation.json"), "w") as f:
json.dump(conversations[i], f, ensure_ascii=False, indent=2)
return results
# ── Item loading (for compatibility with split_three_way) ────────────────────
def load_alfworld_items(
eval_dataset: str,
env_num: int,
seed: int = 42,
is_train: bool = False,
) -> list[dict]:
"""Create pseudo-item dicts for ALFWorld environments.
Since ALFWorld doesn't have a static JSON dataset like SpreadsheetBench,
we create lightweight item dicts that carry enough metadata for the pipeline.
The actual environment is built dynamically.
Returns:
List of dicts with "id" keys, one per environment slot.
"""
items = []
for i in range(env_num):
items.append({
"id": f"env_{i:03d}",
"eval_dataset": eval_dataset,
"env_index": i,
})
return items
+45
View File
@@ -0,0 +1,45 @@
# ALFWorld Embodied Agent Skill
## Overview
This skill guides agents operating in the ALFWorld text-based embodied environment.
The agent must complete household tasks by navigating rooms, interacting with objects,
and using appliances. Actions must be chosen from the admissible action list provided
at each step.
**Output format**: Always output `<think>...</think>` for reasoning, then `<action>...</action>` for the chosen action.
---
## Task Types
| Type | Goal | Key Steps |
|------|------|-----------|
| Pick & Place | Put object X in/on receptacle Y | Find X -> take X -> go to Y -> put X in/on Y |
| Pick Two & Place | Put two instances of X in/on Y | Find X1 -> take -> place -> find X2 -> take -> place |
| Examine in Light | Examine object X under desklamp | Find X -> take X -> find desklamp -> use desklamp |
| Clean & Place | Clean object X and put in/on Y | Find X -> take X -> go to sink -> clean X -> go to Y -> put X |
| Heat & Place | Heat object X and put in/on Y | Find X -> take X -> go to microwave -> heat X -> go to Y -> put X |
| Cool & Place | Cool object X and put in/on Y | Find X -> take X -> go to fridge -> cool X -> go to Y -> put X |
---
## General Principles
1. **Decompose the task**: Parse the goal into ordered sub-goals (locate, acquire, transform, deliver). Complete each before moving to the next.
2. **Systematic exploration**: Search each surface and container exactly once before revisiting. Open closed containers (drawers, cabinets, fridge) before judging them empty.
3. **Grab immediately**: When a required object is visible and reachable, take it right away before moving elsewhere.
4. **Transform before placing**: If the task requires cleaning, heating, or cooling, perform the state change at the appropriate appliance before heading to the final destination.
5. **Direct delivery**: Once holding the transformed (or untransformed) goal object, navigate straight to the target receptacle and place it.
6. **Track progress**: Maintain an internal count of how many objects still need to be found and placed. Only stop searching when the count reaches zero.
7. **Avoid loops**: Never repeat the same action more than twice in a row. If stuck, move to a different unexplored location.
8. **Only choose admissible actions**: Always pick an action from the admissible action list. Do not invent actions.
---
## Common Mistakes to Avoid
- **Revisiting searched locations**: Keep track of which surfaces/containers have been checked; do not re-examine them.
- **Ignoring visible objects**: If the target object appears in the observation, pick it up immediately.
- **Skipping state changes**: Do not place an object at the destination without first cleaning/heating/cooling it when required.
- **Premature termination**: Do not stop the episode until all goal conditions are verified as met.
- **Action loops**: Repeatedly toggling or examining the same object wastes steps. Move on to new locations instead.
+9
View File
@@ -0,0 +1,9 @@
"""Vendored ALFWorld environment runtime.
Minimal subset of SkillRL's agent_system package needed to run
ALFWorld environments with ReflACT. Original source:
https://github.com/NTU-LANTERN/SkillRL (Apache-2.0 License)
"""
from .alfworld_envs import AlfworldEnvs, build_alfworld_envs
from .alfworld_projection import alfworld_projection
from .env_manager import AlfWorldEnvironmentManager
+221
View File
@@ -0,0 +1,221 @@
# Vendored from SkillRL (Apache-2.0 License)
# Original: agent_system/environments/env_package/alfworld/envs.py
# Modified: imports use pip-installed alfworld package instead of vendored copy.
import os
import multiprocessing as mp
import traceback
import yaml
import gymnasium as gym
import numpy as np
from alfworld.agents.environment import get_environment
def load_config_file(path):
assert os.path.exists(path), f"Invalid config file: {path}"
with open(path) as reader:
config = yaml.safe_load(reader)
return config
def compute_reward(info, multi_modal=False):
if multi_modal:
reward = 10.0 * float(info['won']) + float(info['goal_condition_success_rate'])
else:
reward = 10.0 * float(info['won'])
return reward
class AlfworldWorker:
"""Stateful worker that holds one ALFWorld sub-environment."""
def __init__(self, config, seed, base_env, gamefile=None):
if gamefile:
base_env.game_files = [gamefile]
if hasattr(base_env, "num_games"):
base_env.num_games = 1
self.env = base_env.init_env(batch_size=1)
self.env.seed(seed)
def step(self, action):
actions = [action]
obs, scores, dones, infos = self.env.step(actions)
infos['observation_text'] = obs
return obs, scores, dones, infos
def reset(self):
obs, infos = self.env.reset()
infos['observation_text'] = obs
return obs, infos
def _worker_loop(cmd_q, result_q, config, seed, is_train, eval_dataset, gamefile):
"""Run one ALFWorld environment in a child process."""
try:
env_type = config['env']['type']
base_env = get_environment(env_type)(
config,
train_eval='train' if is_train else eval_dataset,
)
worker = AlfworldWorker(config, seed, base_env, gamefile)
result_q.put((True, "ready"))
except BaseException:
result_q.put((False, traceback.format_exc()))
return
while True:
cmd, payload = cmd_q.get()
if cmd == "close":
result_q.put((True, None))
return
try:
if cmd == "reset":
result = worker.reset()
elif cmd == "step":
result = worker.step(payload)
else:
raise ValueError(f"Unknown ALFWorld worker command: {cmd}")
result_q.put((True, result))
except BaseException:
result_q.put((False, traceback.format_exc()))
class _ProcessWorker:
"""Small stdlib actor wrapper for one environment process."""
def __init__(self, ctx, config, seed, is_train, eval_dataset, gamefile=None):
self.cmd_q = ctx.Queue(maxsize=1)
self.result_q = ctx.Queue(maxsize=1)
self.process = ctx.Process(
target=_worker_loop,
args=(self.cmd_q, self.result_q, config, seed, is_train, eval_dataset, gamefile),
)
self.process.start()
ok, payload = self.result_q.get()
if not ok:
self.close(kill=True)
raise RuntimeError(f"Failed to start ALFWorld worker:\n{payload}")
def send(self, cmd, payload=None):
self.cmd_q.put((cmd, payload))
def recv(self):
ok, payload = self.result_q.get()
if not ok:
raise RuntimeError(f"ALFWorld worker failed:\n{payload}")
return payload
def close(self, kill=False):
if self.process.is_alive() and not kill:
try:
self.send("close")
self.recv()
except Exception:
kill = True
if kill and self.process.is_alive():
self.process.terminate()
self.process.join(timeout=5)
if self.process.is_alive():
self.process.kill()
self.process.join(timeout=1)
self.cmd_q.close()
self.result_q.close()
class AlfworldEnvs(gym.Env):
"""Vectorized ALFWorld environment using local process workers."""
def __init__(self, alf_config_path, seed, env_num, group_n,
resources_per_worker, is_train=True, env_kwargs=None, gamefiles=None):
super().__init__()
if env_kwargs is None:
env_kwargs = {}
eval_dataset = env_kwargs.get('eval_dataset', 'eval_in_distribution')
config = load_config_file(alf_config_path)
env_type = config['env']['type']
self.multi_modal = (env_type == 'AlfredThorEnv')
self.num_processes = env_num * group_n
self.group_n = group_n
self.gamefiles = list(gamefiles or [])
if self.gamefiles and len(self.gamefiles) != self.num_processes:
raise ValueError(
f"Expected {self.num_processes} gamefiles, got {len(self.gamefiles)}"
)
start_method = os.environ.get("ALFWORLD_WORKER_START_METHOD") or None
ctx = mp.get_context(start_method) if start_method else mp.get_context()
self.workers = []
for i in range(self.num_processes):
worker_gamefile = self.gamefiles[i] if self.gamefiles else None
worker = _ProcessWorker(
ctx,
config,
seed + (i // self.group_n),
is_train,
eval_dataset,
worker_gamefile,
)
self.workers.append(worker)
self.prev_admissible_commands = [None for _ in range(self.num_processes)]
def step(self, actions):
assert len(actions) == self.num_processes
for i, worker in enumerate(self.workers):
worker.send("step", actions[i])
results = [worker.recv() for worker in self.workers]
text_obs_list = []
rewards_list = []
dones_list = []
info_list = []
for i, (obs, scores, dones, info) in enumerate(results):
for k in info.keys():
info[k] = info[k][0]
text_obs_list.append(obs[0])
dones_list.append(dones[0])
info_list.append(info)
self.prev_admissible_commands[i] = info['admissible_commands']
rewards_list.append(compute_reward(info, self.multi_modal))
image_obs_list = None
return text_obs_list, image_obs_list, rewards_list, dones_list, info_list
def reset(self):
for worker in self.workers:
worker.send("reset")
results = [worker.recv() for worker in self.workers]
text_obs_list = []
info_list = []
for i, (obs, info) in enumerate(results):
for k in info.keys():
info[k] = info[k][0]
text_obs_list.append(obs[0])
self.prev_admissible_commands[i] = info['admissible_commands']
info_list.append(info)
image_obs_list = None
return text_obs_list, image_obs_list, info_list
@property
def get_admissible_commands(self):
return self.prev_admissible_commands
def close(self):
for worker in self.workers:
worker.close()
def build_alfworld_envs(alf_config_path, seed, env_num, group_n,
resources_per_worker, is_train=True, env_kwargs=None, gamefiles=None):
"""Build vectorized ALFWorld environments."""
return AlfworldEnvs(
alf_config_path, seed, env_num, group_n,
resources_per_worker, is_train, env_kwargs, gamefiles,
)
+60
View File
@@ -0,0 +1,60 @@
# Vendored from SkillRL (Apache-2.0 License)
# Original: agent_system/environments/env_package/alfworld/projection.py
from typing import List
import re
def alfworld_projection(actions: List[str], action_pools: List[List[str]]):
"""Process raw model outputs into valid ALFWorld actions.
Extracts text from ``<action>...</action>`` tags and validates that
the response also contains ``<think>...</think>`` tags.
Parameters
----------
actions : list[str]
Raw model outputs, one per environment.
action_pools : list[list[str]]
Admissible action lists per environment (unused but kept for API compat).
Returns
-------
actions : list[str]
Cleaned action strings.
valids : list[int]
1 if the action was successfully parsed, 0 otherwise.
"""
valids = [0] * len(actions)
for i in range(len(actions)):
original_str = actions[i]
actions[i] = actions[i].lower()
start_tag = "<action>"
end_tag = "</action>"
start_idx = actions[i].find(start_tag)
end_idx = actions[i].find(end_tag)
try:
if start_idx == -1 or end_idx == -1:
actions[i] = actions[i][-30:]
continue
extracted_action = actions[i][start_idx + len(start_tag):end_idx].strip().lower()
actions[i] = extracted_action
valids[i] = 1
except Exception:
actions[i] = actions[i][-30:]
# Require <think>...</think>
think_start_idx = original_str.find("<think>")
think_end_idx = original_str.find("</think>")
if think_start_idx == -1 or think_end_idx == -1:
valids[i] = 0
# Reject responses containing Chinese characters
if re.search(r'[\u4e00-\u9fff]', original_str):
valids[i] = 0
return actions, valids
+8
View File
@@ -0,0 +1,8 @@
# Vendored from SkillRL (Apache-2.0 License)
# Original: agent_system/environments/prompts/alfworld.py
from skillopt.prompts import load_prompt
ALFWORLD_TEMPLATE_NO_HIS = load_prompt("rollout_no_history", env="alfworld")
ALFWORLD_TEMPLATE = load_prompt("rollout_with_history", env="alfworld")
ALFWORLD_TEMPLATE_WITH_MEMORY = load_prompt("rollout_with_memory", env="alfworld")
+145
View File
@@ -0,0 +1,145 @@
dataset:
data_path: '$ALFWORLD_DATA/json_2.1.1/train'
eval_id_data_path: '$ALFWORLD_DATA/json_2.1.1/valid_seen' # null/None to disable
eval_ood_data_path: '$ALFWORLD_DATA/json_2.1.1/valid_unseen' # null/None to disable
num_train_games: -1 # max training games (<=0 indicates full dataset)
num_eval_games: -1 # max evaluation games (<=0 indicates full dataset)
logic:
domain: '$ALFWORLD_DATA/logic/alfred.pddl' # PDDL domain file that defines the world dynamics
grammar: '$ALFWORLD_DATA/logic/alfred.twl2' # Grammar file that defines the text feedbacks
env:
type: 'AlfredTWEnv' # 'AlfredTWEnv' or 'AlfredThorEnv' or 'AlfredHybrid'
# regen_game_files: False # check if game is solvable by expert and save to game.tw-pddl file
domain_randomization: False # shuffle Textworld print order and object id nums
task_types: [1, 2, 3, 4, 5, 6] # task-type ids: 1 - Pick & Place, 2 - Examine in Light, 3 - Clean & Place, 4 - Heat & Place, 5 - Cool & Place, 6 - Pick Two & Place
expert_timeout_steps: 150 # max steps before timeout for expert to solve the task
expert_type: "handcoded" # 'handcoded' or 'planner'. Note: the planner is very slow for real-time use
goal_desc_human_anns_prob: 0.0 # prob of using human-annotated goal language instead of templated goals (1.0 indicates all human annotations from ALFRED)
hybrid:
start_eps: 100000 # starting episode of hybrid training, tw-only training upto this point
thor_prob: 0.5 # prob of AlfredThorEnv during hybrid training
eval_mode: "tw" # 'tw' or 'thor' - env used for evaluation during hybrid training
thor:
screen_width: 300 # width of THOR window
screen_height: 300 # height of THOR window
smooth_nav: False # smooth rotations, looks, and translations during navigation (very slow)
save_frames_to_disk: False # save frame PNGs to disk (useful for making videos)
save_frames_path: './videos/' # path to save frame PNGs
controller:
type: 'oracle' # 'oracle' or 'oracle_astar' or 'mrcnn' or 'mrcnn_astar' (aka BUTLER)
debug: False
load_receps: True # load receptacle locations from precomputed dict (if available)
mask_rcnn:
pretrained_model_path: '$ALFWORLD_DATA/detectors/mrcnn.pth'
general:
random_seed: 42
use_cuda: True # disable this when running on machine without cuda
visdom: False # plot training/eval curves, run with visdom server
task: 'alfred'
training_method: 'dagger' # 'dqn' or 'dagger'
save_path: './training/' # path to save pytorch models
observation_pool_capacity: 3 # k-size queue, 0 indicates no observation
hide_init_receptacles: False # remove initial observation containing navigable receptacles
training:
batch_size: 10
max_episode: 50000
smoothing_eps: 0.1
optimizer:
learning_rate: 0.001
clip_grad_norm: 5
evaluate:
run_eval: True
batch_size: 10
env:
type: "AlfredTWEnv"
checkpoint:
report_frequency: 1000 # report every N episode
experiment_tag: 'test' # name of experiment
load_pretrained: False # during test, enable this so that the agent load your pretrained model
load_from_tag: 'not loading anything' # name of pre-trained model to load in save_path
model:
encoder_layers: 1
decoder_layers: 1
encoder_conv_num: 5
block_hidden_dim: 64
n_heads: 1
dropout: 0.1
block_dropout: 0.1
recurrent: True
rl:
action_space: "admissible" # 'admissible' (candidates from text engine) or 'generation' (seq2seq-style generation) or 'beam_search_choice' or 'exhaustive' (not working)
max_target_length: 20 # max token length for seq2seq generation
beam_width: 10 # 1 means greedy
generate_top_k: 3
training:
max_nb_steps_per_episode: 50 # terminate after this many steps
learn_start_from_this_episode: 0 # delay updates until this epsiode
target_net_update_frequency: 500 # sync target net with online net per this many epochs
replay:
accumulate_reward_from_final: True
count_reward_lambda: 0.0 # 0 to disable
novel_object_reward_lambda: 0.0 # 0 to disable
discount_gamma_game_reward: 0.9
discount_gamma_count_reward: 0.5
discount_gamma_novel_object_reward: 0.5
replay_memory_capacity: 500000 # adjust this depending on your RAM size
replay_memory_priority_fraction: 0.5
update_per_k_game_steps: 5
replay_batch_size: 64
multi_step: 3
replay_sample_history_length: 4
replay_sample_update_from: 2
epsilon_greedy:
noisy_net: False # if this is true, then epsilon greedy is disabled
epsilon_anneal_episodes: 1000 # -1 if not annealing
epsilon_anneal_from: 0.3
epsilon_anneal_to: 0.1
dagger:
action_space: "generation" # 'admissible' (candidates from text engine) or 'generation' (seq2seq-style generation) or 'exhaustive' (not working)
max_target_length: 20 # max token length for seq2seq generation
beam_width: 10 # 1 means greedy
generate_top_k: 5
unstick_by_beam_search: False # use beam-search for failed actions, set True during evaluation
training:
max_nb_steps_per_episode: 50 # terminate after this many steps
fraction_assist:
fraction_assist_anneal_episodes: 50000
fraction_assist_anneal_from: 1.0
fraction_assist_anneal_to: 0.01
fraction_random:
fraction_random_anneal_episodes: 0
fraction_random_anneal_from: 0.0
fraction_random_anneal_to: 0.0
replay:
replay_memory_capacity: 500000
update_per_k_game_steps: 5
replay_batch_size: 64
replay_sample_history_length: 4
replay_sample_update_from: 2
vision_dagger:
model_type: "resnet" # 'resnet' (whole image features) or 'maskrcnn_whole' (whole image MaskRCNN feats) or 'maskrcnn' (top k MaskRCNN detection feats) or 'no_vision' (zero vision input)
resnet_fc_dim: 64
maskrcnn_top_k_boxes: 10 # top k box features
use_exploration_frame_feats: False # append feats from initial exploration (memory intensive!)
sequence_aggregation_method: "average" # 'sum' or 'average' or 'rnn'
+84
View File
@@ -0,0 +1,84 @@
# Vendored from SkillRL (Apache-2.0 License)
# Original: agent_system/environments/base.py
# Trimmed to only include what ALFWorld needs.
from typing import List, Tuple, Dict, Any
import numpy as np
from collections import defaultdict
def to_numpy(data):
"""Convert data to numpy array."""
# Lazy-check for torch.Tensor to avoid hard dependency on torch
_torch_tensor = None
try:
import torch
_torch_tensor = torch.Tensor
except ImportError:
pass
if _torch_tensor is not None and isinstance(data, _torch_tensor):
data = data.detach().cpu().numpy()
elif isinstance(data, np.ndarray):
pass
elif isinstance(data, (int, float, bool, Tuple, List)):
data = np.array(data)
else:
raise ValueError(f"Unsupported type: {type(data)})")
return data
class EnvironmentManagerBase:
"""Base class for vectorized environment managers.
Manages a set of parallel environments, handles action projection,
observation post-processing, and history tracking.
"""
def __init__(self, envs, projection_f, config):
self.envs = envs
self.projection_f = projection_f
self.config = config
def reset(self, kwargs) -> Dict[str, Any]:
obs, infos = self.envs.reset()
return {'text': None, 'image': obs, 'anchor': None}, infos
def step(self, text_actions: List[str]):
actions, valids = self.projection_f(text_actions)
next_obs, rewards, dones, infos = self.envs.step(actions)
next_observations = {
'text': None,
'image': next_obs,
'anchor': None,
}
for i, info in enumerate(infos):
info['is_action_valid'] = to_numpy(valids[i])
rewards = to_numpy(rewards)
dones = to_numpy(dones)
return next_observations, rewards, dones, infos
def close(self) -> None:
self.envs.close()
def success_evaluator(self, *args, **kwargs) -> Dict[str, np.ndarray]:
total_infos = kwargs['total_infos']
total_batch_list = kwargs['total_batch_list']
batch_size = len(total_batch_list)
success = defaultdict(list)
for bs in range(batch_size):
self._process_batch(bs, total_batch_list, total_infos, success)
assert len(success['success_rate']) == batch_size
return {key: np.array(value) for key, value in success.items()}
def _process_batch(self, batch_idx, total_batch_list, total_infos, success):
for i in reversed(range(len(total_batch_list[batch_idx]))):
batch_item = total_batch_list[batch_idx][i]
if batch_item['active_masks']:
info = total_infos[batch_idx][i]
won_value = float(info['won'])
success['success_rate'].append(won_value)
return
+139
View File
@@ -0,0 +1,139 @@
# Vendored from SkillRL (Apache-2.0 License)
# Original: agent_system/environments/env_manager.py
# Trimmed to only include AlfWorldEnvironmentManager and its helpers.
from typing import List, Dict, Any
from collections import defaultdict
import numpy as np
from skillopt.envs.alfworld.vendor.env_base import EnvironmentManagerBase, to_numpy
from skillopt.envs.alfworld.vendor.alfworld_prompts import (
ALFWORLD_TEMPLATE,
ALFWORLD_TEMPLATE_NO_HIS,
ALFWORLD_TEMPLATE_WITH_MEMORY,
)
from skillopt.envs.alfworld.vendor.memory import SimpleMemory
def parse_gamefile(infos):
gamefile = []
for info in infos:
if 'extra.gamefile' in info:
gamefile.append(info['extra.gamefile'])
else:
gamefile.append(None)
return gamefile
def set_gamefile(infos, gamefile):
for i in range(len(infos)):
if 'extra.gamefile' in infos[i]:
infos[i]['extra.gamefile'] = gamefile[i]
else:
infos[i]['extra.gamefile'] = None
return infos
class AlfWorldEnvironmentManager(EnvironmentManagerBase):
"""Manages parallel ALFWorld environments with observation templating."""
def __init__(self, envs, projection_f, config):
self.memory = SimpleMemory()
self.retrieval_memory = None
super().__init__(envs, projection_f, config)
def reset(self, kwargs):
text_obs, image_obs, infos = self.envs.reset()
self.gamefile = parse_gamefile(infos)
self.memory.reset(batch_size=len(text_obs))
self.tasks = []
self.pre_text_obs = text_obs
self.extract_task(text_obs)
full_text_obs = self.build_text_obs(text_obs, self.envs.get_admissible_commands, init=True)
return {'text': full_text_obs, 'image': image_obs, 'anchor': text_obs}, infos
def step(self, text_actions: List[str]):
actions, valids = self.projection_f(text_actions, self.envs.get_admissible_commands)
text_obs, image_obs, rewards, dones, infos = self.envs.step(actions)
self.memory.store({'text_obs': self.pre_text_obs, 'action': actions})
self.pre_text_obs = text_obs
full_text_obs = self.build_text_obs(text_obs, self.envs.get_admissible_commands)
if infos[0].get("extra.gamefile") is None:
infos = set_gamefile(infos, self.gamefile)
for i, info in enumerate(infos):
info['is_action_valid'] = to_numpy(valids[i])
next_observations = {'text': full_text_obs, 'image': image_obs, 'anchor': text_obs}
rewards = to_numpy(rewards)
dones = to_numpy(dones)
return next_observations, rewards, dones, infos
def extract_task(self, text_obs: List[str]):
for obs in text_obs:
task_start = obs.find('Your task is to: ')
if task_start != -1:
self.tasks.append(obs[task_start + len('Your task is to: '):].strip())
else:
raise ValueError("Task description not found in text observation.")
def build_text_obs(self, text_obs: List[str], admissible_actions: List[List[str]], init: bool = False) -> List[str]:
postprocess_text_obs = []
if not init and self.config.env.history_length > 0:
memory_contexts, valid_lens = self.memory.fetch(
self.config.env.history_length,
obs_key="text_obs",
action_key="action",
)
for i in range(len(text_obs)):
reformatted_admissible_actions = "\n ".join(
f"'{s}'" for s in admissible_actions[i] if s != 'help'
)
if init or self.config.env.history_length <= 0:
obs = ALFWORLD_TEMPLATE_NO_HIS.format(
current_observation=text_obs[i],
admissible_actions=reformatted_admissible_actions,
)
else:
obs = ALFWORLD_TEMPLATE.format(
task_description=self.tasks[i],
step_count=len(self.memory[i]),
history_length=valid_lens[i],
action_history=memory_contexts[i],
current_step=len(self.memory[i]) + 1,
current_observation=text_obs[i],
admissible_actions=reformatted_admissible_actions,
)
postprocess_text_obs.append(obs)
return postprocess_text_obs
def _process_batch(self, batch_idx, total_batch_list, total_infos, success):
for i in reversed(range(len(total_batch_list[batch_idx]))):
batch_item = total_batch_list[batch_idx][i]
if batch_item['active_masks']:
info = total_infos[batch_idx][i]
won_value = float(info['won'])
success['success_rate'].append(won_value)
gamefile = info.get("extra.gamefile")
if gamefile:
self._process_gamefile(gamefile, won_value, success)
return
def _process_gamefile(self, gamefile, won_value, success):
tasks = [
"pick_and_place",
"pick_two_obj_and_place",
"look_at_obj_in_light",
"pick_heat_then_place_in_recep",
"pick_cool_then_place_in_recep",
"pick_clean_then_place_in_recep",
]
for task in tasks:
if task in gamefile:
success[f"{task}_success_rate"].append(won_value)
break
+87
View File
@@ -0,0 +1,87 @@
# Vendored from SkillRL (Apache-2.0 License)
# Original: agent_system/memory/base.py + agent_system/memory/memory.py
# Merged into a single file for simplicity.
from abc import ABC, abstractmethod
from typing import List, Dict, Any, Tuple
class BaseMemory(ABC):
"""Base class for memory management."""
@abstractmethod
def __len__(self):
pass
@abstractmethod
def __getitem__(self, idx: int):
pass
@abstractmethod
def reset(self, batch_size: int):
pass
@abstractmethod
def store(self, record: Dict[str, List[Any]]):
pass
@abstractmethod
def fetch(self, step: int):
pass
class SimpleMemory(BaseMemory):
"""Per-environment history buffer for storing observations and actions."""
def __init__(self):
self._data = None
self.keys = None
self.batch_size = 0
def __len__(self):
return len(self._data)
def __getitem__(self, idx):
return self._data[idx]
def reset(self, batch_size: int):
if self._data is not None:
self._data.clear()
self._data = [[] for _ in range(batch_size)]
self.batch_size = batch_size
self.keys = None
def store(self, record: Dict[str, List[Any]]):
if self.keys is None:
self.keys = list(record.keys())
assert self.keys == list(record.keys())
for env_idx in range(self.batch_size):
self._data[env_idx].append({k: record[k][env_idx] for k in self.keys})
def fetch(
self,
history_length: int,
obs_key: str = "text_obs",
action_key: str = "action",
) -> Tuple[List[str], List[int]]:
memory_contexts, valid_lengths = [], []
for env_idx in range(self.batch_size):
recent = self._data[env_idx][-history_length:]
valid_len = len(recent)
start_idx = len(self._data[env_idx]) - valid_len
lines = []
for j, rec in enumerate(recent):
step_num = start_idx + j + 1
act = rec[action_key]
obs = rec[obs_key]
lines.append(
f"[Observation {step_num}: '{obs}', Action {step_num}: '{act}']"
)
memory_contexts.append("\n".join(lines))
valid_lengths.append(valid_len)
return memory_contexts, valid_lengths