ysharma's picture
ysharma HF Staff
fix frame timing: sample rate is 50/every, so video played 0.8x and the telemetry time axis was 25% long
916a9db verified
Raw History Blame Contribute Delete
13.9 kB
"""Turn a routine (a list of moves) into a rollout of the real policies.
A plan is a JSON list of steps, e.g.
[{"move": "forward", "seconds": 2.5},
{"move": "sit", "seconds": 2.0},
{"move": "kick_right"},
{"move": "skate_forward", "seconds": 3.0}]
Duration-based moves take `seconds`; the one-shots (`kick_*`, `roll`, `pick`,
`skate_crouch`) run for exactly as long as their trained cycle lasts, so
`seconds` is ignored for them.
`skate_*` moves need the roller robot, which is a different MJCF. Switching
locomotion rebuilds the model and respawns the duck at the origin, exactly as
the official playground does.
"""
import json
import numpy as np
import duck as D
# move -> (locomotion variant, policy mode, command [vx, vy, wz])
# Velocities are filled in per variant at run time from sim.vel_limits().
MOVES = {
# legs
"forward": ("legs", "walk", (1.0, 0.0, 0.0)),
"backward": ("legs", "walk", (-1.0, 0.0, 0.0)),
"strafe_left": ("legs", "walk", (0.0, 0.6, 0.0)),
"strafe_right": ("legs", "walk", (0.0, -0.6, 0.0)),
"turn_left": ("legs", "walk", (0.0, 0.0, 1.0)),
"turn_right": ("legs", "walk", (0.0, 0.0, -1.0)),
"stand": ("legs", "walk", (0.0, 0.0, 0.0)),
"sit": ("legs", "sitstand", (0.0, 0.0, 0.0)),
"kick_left": ("legs", "kickL", (0.0, 0.0, 0.0)),
"kick_right": ("legs", "kickR", (0.0, 0.0, 0.0)),
"roll": ("legs", "roll", (0.0, 0.0, 0.0)),
"pick": ("legs", "groundpick", (0.0, 0.0, 0.0)),
# head gestures (cmd[3..6]); the duck holds still and moves its head
"look_up": ("legs", "walk", (0.0, 0.0, 0.0)),
"look_down": ("legs", "walk", (0.0, 0.0, 0.0)),
"look_left": ("legs", "walk", (0.0, 0.0, 0.0)),
"look_right": ("legs", "walk", (0.0, 0.0, 0.0)),
"tilt_head": ("legs", "walk", (0.0, 0.0, 0.0)),
# rollers
"skate_forward": ("rollers", "walk", (1.0, 0.0, 0.0)),
"skate_backward": ("rollers", "walk", (-1.0, 0.0, 0.0)),
"skate_turn_left": ("rollers", "walk", (0.0, 0.0, 1.0)),
"skate_turn_right": ("rollers", "walk", (0.0, 0.0, -1.0)),
"skate_stand": ("rollers", "walk", (0.0, 0.0, 0.0)),
"skate_crouch": ("rollers", "crouch", (0.0, 0.0, 0.0)),
}
# Once the skates are on they stay on: a legs move appearing AFTER the first
# skate move is the LLM losing track, not a deliberate change of body, and
# honouring it would swap the model and respawn the duck mid-routine. Moves
# with a roller equivalent are coerced; the rest are dropped with a note.
# A legs move BEFORE any skating is left alone, so "walk, then skate" works.
TO_ROLLER = {
"forward": "skate_forward", "backward": "skate_backward",
"turn_left": "skate_turn_left", "turn_right": "skate_turn_right",
"strafe_left": "skate_turn_left", "strafe_right": "skate_turn_right",
"stand": "skate_stand", "roll": "skate_crouch",
}
ONE_SHOT = {"kick_left", "kick_right", "roll", "pick", "skate_crouch"}
# Head pose per gesture, as a fraction of HEAD_MAX:
# [neck_pitch, head_pitch, head_yaw, head_roll]. Signs follow game.js's
# HEAD_SIGNS so "up" reads as up.
HEAD_POSES = {
"look_up": (0.55, -0.45, 0.0, 0.0),
"look_down": (-0.5, 0.5, 0.0, 0.0),
"look_left": (0.0, 0.0, 0.6, 0.0),
"look_right": (0.0, 0.0, -0.6, 0.0),
"tilt_head": (0.15, 0.0, 0.15, -0.6),
}
MAX_SECONDS = 30.0
# Loose spellings the LLM (or a human) may produce.
ALIASES = {
"walk": "forward", "walk_forward": "forward", "forwards": "forward",
"waddle": "forward",
"back": "backward", "backwards": "backward", "reverse": "backward",
"left": "turn_left", "right": "turn_right",
"rotate_left": "turn_left", "rotate_right": "turn_right",
"spin": "turn_left", "spin_left": "turn_left", "spin_right": "turn_right",
"sidestep_left": "strafe_left", "sidestep_right": "strafe_right",
"wait": "stand", "idle": "stand", "pause": "stand",
"sit_down": "sit", "sitting": "sit", "rest": "sit", "crouch_down": "sit",
"stand_up": "stand", "get_up": "stand",
"kick": "kick_right", "shoot": "kick_right",
"roulade": "roll", "somersault": "roll", "flip": "roll", "tumble": "roll",
"pickup": "pick", "pick_up": "pick", "grab": "pick", "fetch": "pick",
"look": "look_up", "look_around": "look_left", "nod": "look_down",
"head_tilt": "tilt_head", "tilt": "tilt_head", "curious": "tilt_head",
"skate": "skate_forward", "roller_skate": "skate_forward",
"rollerblade": "skate_forward", "skating": "skate_forward",
"skate_left": "skate_turn_left", "skate_right": "skate_turn_right",
"glide": "skate_crouch", "skate_glide": "skate_crouch",
}
def parse_plan(text):
"""Parse a plan from JSON text. Returns (steps, notes).
Tolerant on purpose: the plan usually comes from an LLM. Unknown moves are
dropped and reported in `notes` rather than raising.
"""
notes = []
raw = (text or "").strip()
if not raw:
return [], ["empty plan"]
# Tolerate ```json fences and any prose around the array.
if "```" in raw:
raw = raw.split("```")[1]
if raw.lstrip().lower().startswith("json"):
raw = raw.lstrip()[4:]
start, end = raw.find("["), raw.rfind("]")
if start != -1 and end > start:
raw = raw[start:end + 1]
try:
parsed = json.loads(raw)
except json.JSONDecodeError as e:
return [], ["could not parse plan as JSON: " + str(e)]
if isinstance(parsed, dict):
parsed = parsed.get("plan") or parsed.get("steps") or [parsed]
if not isinstance(parsed, list):
return [], ["plan is not a list"]
steps, total = [], 0.0
for item in parsed:
if isinstance(item, str):
item = {"move": item}
if not isinstance(item, dict):
continue
name = str(item.get("move") or item.get("action") or "").strip().lower()
name = name.replace("-", "_").replace(" ", "_")
name = ALIASES.get(name, name)
if name not in MOVES:
notes.append("skipped unknown move " + repr(name))
continue
if name in ONE_SHOT:
secs = 0.0
else:
try:
secs = float(item.get("seconds", item.get("duration", 1.5)))
except (TypeError, ValueError):
secs = 1.5
secs = float(np.clip(secs, 0.2, 8.0))
if total + max(secs, 3.0) > MAX_SECONDS:
notes.append("plan truncated at " + str(MAX_SECONDS) + "s")
break
total += secs if secs else 3.0
steps.append({"move": name, "seconds": secs})
steps, coerce_notes = _keep_skates_on(steps)
notes.extend(coerce_notes)
if not steps:
notes.append("no runnable moves found")
return steps, notes
def _keep_skates_on(steps):
"""After the first skate move, hold the routine on the roller body."""
first = next((i for i, s in enumerate(steps)
if MOVES[s["move"]][0] == "rollers"), None)
if first is None:
return steps, []
out, notes = steps[:first], []
for s in steps[first:]:
name = s["move"]
if MOVES[name][0] == "rollers":
out.append(s)
elif name in TO_ROLLER:
swapped = TO_ROLLER[name]
notes.append(name + " -> " + swapped + " (still on skates)")
out.append({"move": swapped,
"seconds": 0.0 if swapped in ONE_SHOT else s["seconds"]})
else:
notes.append("dropped " + name + " (no skating equivalent)")
return out, notes
def run(plan_steps, seed=0, width=640, height=360, fps=20, on_frame=None):
"""Roll the plan out through the real policies.
Returns (frames, telemetry). Locomotion variants are built lazily and
cached, so a legs-only routine never pays for the roller model.
"""
rng = np.random.default_rng(seed)
# Frames can only be sampled on whole control steps, so the achievable
# rate is 50/every, not whatever was asked for. Derive it back or the
# video plays at the wrong speed and the telemetry time axis is stretched
# (requesting 20 still samples every 2nd step = 25 fps, so the video ran
# at 0.8x and an 11 s routine charted as 14 s).
ctrl_hz = 1.0 / D.CTRL_DT
every = max(1, int(round(ctrl_hz / max(fps, 1e-6))))
fps = ctrl_hz / every
sims = {}
sim = None
frames, track, upright, modes = [], [], [], []
log, notes = [], []
tick = 0
def use(variant):
"""Activate a locomotion variant, building it on first use."""
nonlocal sim
if sim is not None and sim.variant == variant:
return False
if variant not in sims:
sims[variant] = D.Microduck(width=width, height=height,
render=on_frame is None, variant=variant)
sim = sims[variant]
sim.reset()
return True
def do_step(mode, cmd, phase=None, sit_flag=0.0):
nonlocal tick
sim.control_step(mode, cmd, phase, sit_flag)
if tick % every == 0:
img = sim.frame() if on_frame is None else on_frame(sim)
if img is not None:
frames.append(img)
track.append((float(sim.data.qpos[0]), float(sim.data.qpos[1])))
upright.append(float(sim.proj_gravity()[2]))
modes.append("recover" if sim.recovery else mode)
tick += 1
def settle(min_steps=25, max_steps=400):
"""Hand back to the locomotion policy, and if the move ended on the
floor let the get-up policy finish before the next move starts."""
for n in range(max_steps):
do_step("walk", (0.0, 0.0, 0.0))
if n >= min_steps and sim.recovery is None:
return
def scaled(cmd):
"""Turn the unit command into this variant's real velocity limits."""
fwd, back, ang = sim.vel_limits()
vx = cmd[0] * (fwd if cmd[0] >= 0 else -back)
return (vx, cmd[1], cmd[2] * ang)
# No eager build: a skate-only routine never pays for the legs model.
for step in plan_steps:
name = step["move"]
variant, mode, unit_cmd = MOVES[name]
switched = use(variant)
if switched and log:
notes.append("switched to " + variant + " at " + name
+ " (robot respawns at the origin)")
cmd = scaled(unit_cmd)
t0 = tick * D.CTRL_DT
x0, y0 = float(sim.data.qpos[0]), float(sim.data.qpos[1])
# Head gestures persist until changed, like the runtime's offsets.
if name in HEAD_POSES:
sim.head_target[:] = np.array(HEAD_POSES[name], dtype=np.float32) * D.HEAD_MAX
elif name not in ONE_SHOT and name != "sit":
sim.head_target[:] = 0.0
if name in ("kick_left", "kick_right"):
# Plant first: the ball is placed relative to the duck's stance,
# so a kick entered mid-stride would put it under a moving foot.
for _ in range(25):
do_step("walk", (0.0, 0.0, 0.0))
sim.spawn_ball(rng, foot="L" if name == "kick_left" else "R")
for _ in range(D.KICK_STEPS):
do_step(mode, cmd)
sim.post_kick_lock = D.POST_KICK_LOCK_STEPS
for _ in range(D.POST_KICK_LOCK_STEPS):
do_step("walk", (0.0, 0.0, 0.0))
elif name == "sit":
# game.js: hold the stand under sitstand for 0.8 s, command the
# sit, then give it 2.0 s to stand back up before moving on.
for _ in range(int(D.SIT_HANDOVER_S / D.CTRL_DT)):
do_step("sitstand", cmd, sit_flag=0.0)
for _ in range(int(round(step["seconds"] / D.CTRL_DT))):
do_step("sitstand", cmd, sit_flag=1.0)
for _ in range(int(D.SIT_STANDUP_S / D.CTRL_DT)):
do_step("sitstand", cmd, sit_flag=0.0)
sim.last_action[:] = 0
elif name == "roll":
tipped, n = False, 0
while n < 150:
do_step(mode, cmd)
n += 1
gz = float(sim.proj_gravity()[2])
if gz > -0.3:
tipped = True
if tipped and gz < -0.85 and n >= 40:
break
sim.last_action[:] = 0
settle()
elif name in ("pick", "skate_crouch"):
period = (D.GROUND_PICK_PERIOD_S if name == "pick"
else D.CROUCH_PERIOD_S)
end = (D.GROUND_PICK_END_PHASE if name == "pick"
else D.CROUCH_END_PHASE)
phase, n = 0.0, 0
while phase < end and n < 400:
do_step(mode, cmd, phase=phase)
phase += D.CTRL_DT / period
n += 1
settle()
else:
for _ in range(int(round(step["seconds"] / D.CTRL_DT))):
do_step(mode, cmd)
log.append({
"move": name,
"t_start": round(t0, 2),
"t_end": round(tick * D.CTRL_DT, 2),
"travelled_m": round(float(np.hypot(sim.data.qpos[0] - x0,
sim.data.qpos[1] - y0)), 3),
"variant": variant,
})
telemetry = {
"track": track,
"upright": upright,
"modes": modes,
"log": log,
"notes": notes,
"duration_s": round(tick * D.CTRL_DT, 2),
"fps": fps,
"falls": sum(1 for a, b in zip(["walk"] + modes, modes)
if b == "recover" and a != "recover"),
"distance_m": round(float(np.hypot(sim.data.qpos[0], sim.data.qpos[1])), 3),
"final_upright": round(float(sim.proj_gravity()[2]), 3),
"variants": sorted({e["variant"] for e in log}),
}
return frames, telemetry