"""Turn a routine (a list of moves) into a rollout of the real policies. A plan is a JSON list of steps, e.g. [{"move": "forward", "seconds": 2.5}, {"move": "sit", "seconds": 2.0}, {"move": "kick_right"}, {"move": "skate_forward", "seconds": 3.0}] Duration-based moves take `seconds`; the one-shots (`kick_*`, `roll`, `pick`, `skate_crouch`) run for exactly as long as their trained cycle lasts, so `seconds` is ignored for them. `skate_*` moves need the roller robot, which is a different MJCF. Switching locomotion rebuilds the model and respawns the duck at the origin, exactly as the official playground does. """ import json import numpy as np import duck as D # move -> (locomotion variant, policy mode, command [vx, vy, wz]) # Velocities are filled in per variant at run time from sim.vel_limits(). MOVES = { # legs "forward": ("legs", "walk", (1.0, 0.0, 0.0)), "backward": ("legs", "walk", (-1.0, 0.0, 0.0)), "strafe_left": ("legs", "walk", (0.0, 0.6, 0.0)), "strafe_right": ("legs", "walk", (0.0, -0.6, 0.0)), "turn_left": ("legs", "walk", (0.0, 0.0, 1.0)), "turn_right": ("legs", "walk", (0.0, 0.0, -1.0)), "stand": ("legs", "walk", (0.0, 0.0, 0.0)), "sit": ("legs", "sitstand", (0.0, 0.0, 0.0)), "kick_left": ("legs", "kickL", (0.0, 0.0, 0.0)), "kick_right": ("legs", "kickR", (0.0, 0.0, 0.0)), "roll": ("legs", "roll", (0.0, 0.0, 0.0)), "pick": ("legs", "groundpick", (0.0, 0.0, 0.0)), # head gestures (cmd[3..6]); the duck holds still and moves its head "look_up": ("legs", "walk", (0.0, 0.0, 0.0)), "look_down": ("legs", "walk", (0.0, 0.0, 0.0)), "look_left": ("legs", "walk", (0.0, 0.0, 0.0)), "look_right": ("legs", "walk", (0.0, 0.0, 0.0)), "tilt_head": ("legs", "walk", (0.0, 0.0, 0.0)), # rollers "skate_forward": ("rollers", "walk", (1.0, 0.0, 0.0)), "skate_backward": ("rollers", "walk", (-1.0, 0.0, 0.0)), "skate_turn_left": ("rollers", "walk", (0.0, 0.0, 1.0)), "skate_turn_right": ("rollers", "walk", (0.0, 0.0, -1.0)), "skate_stand": ("rollers", "walk", (0.0, 0.0, 0.0)), "skate_crouch": ("rollers", "crouch", (0.0, 0.0, 0.0)), } # Once the skates are on they stay on: a legs move appearing AFTER the first # skate move is the LLM losing track, not a deliberate change of body, and # honouring it would swap the model and respawn the duck mid-routine. Moves # with a roller equivalent are coerced; the rest are dropped with a note. # A legs move BEFORE any skating is left alone, so "walk, then skate" works. TO_ROLLER = { "forward": "skate_forward", "backward": "skate_backward", "turn_left": "skate_turn_left", "turn_right": "skate_turn_right", "strafe_left": "skate_turn_left", "strafe_right": "skate_turn_right", "stand": "skate_stand", "roll": "skate_crouch", } ONE_SHOT = {"kick_left", "kick_right", "roll", "pick", "skate_crouch"} # Head pose per gesture, as a fraction of HEAD_MAX: # [neck_pitch, head_pitch, head_yaw, head_roll]. Signs follow game.js's # HEAD_SIGNS so "up" reads as up. HEAD_POSES = { "look_up": (0.55, -0.45, 0.0, 0.0), "look_down": (-0.5, 0.5, 0.0, 0.0), "look_left": (0.0, 0.0, 0.6, 0.0), "look_right": (0.0, 0.0, -0.6, 0.0), "tilt_head": (0.15, 0.0, 0.15, -0.6), } MAX_SECONDS = 30.0 # Loose spellings the LLM (or a human) may produce. ALIASES = { "walk": "forward", "walk_forward": "forward", "forwards": "forward", "waddle": "forward", "back": "backward", "backwards": "backward", "reverse": "backward", "left": "turn_left", "right": "turn_right", "rotate_left": "turn_left", "rotate_right": "turn_right", "spin": "turn_left", "spin_left": "turn_left", "spin_right": "turn_right", "sidestep_left": "strafe_left", "sidestep_right": "strafe_right", "wait": "stand", "idle": "stand", "pause": "stand", "sit_down": "sit", "sitting": "sit", "rest": "sit", "crouch_down": "sit", "stand_up": "stand", "get_up": "stand", "kick": "kick_right", "shoot": "kick_right", "roulade": "roll", "somersault": "roll", "flip": "roll", "tumble": "roll", "pickup": "pick", "pick_up": "pick", "grab": "pick", "fetch": "pick", "look": "look_up", "look_around": "look_left", "nod": "look_down", "head_tilt": "tilt_head", "tilt": "tilt_head", "curious": "tilt_head", "skate": "skate_forward", "roller_skate": "skate_forward", "rollerblade": "skate_forward", "skating": "skate_forward", "skate_left": "skate_turn_left", "skate_right": "skate_turn_right", "glide": "skate_crouch", "skate_glide": "skate_crouch", } def parse_plan(text): """Parse a plan from JSON text. Returns (steps, notes). Tolerant on purpose: the plan usually comes from an LLM. Unknown moves are dropped and reported in `notes` rather than raising. """ notes = [] raw = (text or "").strip() if not raw: return [], ["empty plan"] # Tolerate ```json fences and any prose around the array. if "```" in raw: raw = raw.split("```")[1] if raw.lstrip().lower().startswith("json"): raw = raw.lstrip()[4:] start, end = raw.find("["), raw.rfind("]") if start != -1 and end > start: raw = raw[start:end + 1] try: parsed = json.loads(raw) except json.JSONDecodeError as e: return [], ["could not parse plan as JSON: " + str(e)] if isinstance(parsed, dict): parsed = parsed.get("plan") or parsed.get("steps") or [parsed] if not isinstance(parsed, list): return [], ["plan is not a list"] steps, total = [], 0.0 for item in parsed: if isinstance(item, str): item = {"move": item} if not isinstance(item, dict): continue name = str(item.get("move") or item.get("action") or "").strip().lower() name = name.replace("-", "_").replace(" ", "_") name = ALIASES.get(name, name) if name not in MOVES: notes.append("skipped unknown move " + repr(name)) continue if name in ONE_SHOT: secs = 0.0 else: try: secs = float(item.get("seconds", item.get("duration", 1.5))) except (TypeError, ValueError): secs = 1.5 secs = float(np.clip(secs, 0.2, 8.0)) if total + max(secs, 3.0) > MAX_SECONDS: notes.append("plan truncated at " + str(MAX_SECONDS) + "s") break total += secs if secs else 3.0 steps.append({"move": name, "seconds": secs}) steps, coerce_notes = _keep_skates_on(steps) notes.extend(coerce_notes) if not steps: notes.append("no runnable moves found") return steps, notes def _keep_skates_on(steps): """After the first skate move, hold the routine on the roller body.""" first = next((i for i, s in enumerate(steps) if MOVES[s["move"]][0] == "rollers"), None) if first is None: return steps, [] out, notes = steps[:first], [] for s in steps[first:]: name = s["move"] if MOVES[name][0] == "rollers": out.append(s) elif name in TO_ROLLER: swapped = TO_ROLLER[name] notes.append(name + " -> " + swapped + " (still on skates)") out.append({"move": swapped, "seconds": 0.0 if swapped in ONE_SHOT else s["seconds"]}) else: notes.append("dropped " + name + " (no skating equivalent)") return out, notes def run(plan_steps, seed=0, width=640, height=360, fps=20, on_frame=None): """Roll the plan out through the real policies. Returns (frames, telemetry). Locomotion variants are built lazily and cached, so a legs-only routine never pays for the roller model. """ rng = np.random.default_rng(seed) # Frames can only be sampled on whole control steps, so the achievable # rate is 50/every, not whatever was asked for. Derive it back or the # video plays at the wrong speed and the telemetry time axis is stretched # (requesting 20 still samples every 2nd step = 25 fps, so the video ran # at 0.8x and an 11 s routine charted as 14 s). ctrl_hz = 1.0 / D.CTRL_DT every = max(1, int(round(ctrl_hz / max(fps, 1e-6)))) fps = ctrl_hz / every sims = {} sim = None frames, track, upright, modes = [], [], [], [] log, notes = [], [] tick = 0 def use(variant): """Activate a locomotion variant, building it on first use.""" nonlocal sim if sim is not None and sim.variant == variant: return False if variant not in sims: sims[variant] = D.Microduck(width=width, height=height, render=on_frame is None, variant=variant) sim = sims[variant] sim.reset() return True def do_step(mode, cmd, phase=None, sit_flag=0.0): nonlocal tick sim.control_step(mode, cmd, phase, sit_flag) if tick % every == 0: img = sim.frame() if on_frame is None else on_frame(sim) if img is not None: frames.append(img) track.append((float(sim.data.qpos[0]), float(sim.data.qpos[1]))) upright.append(float(sim.proj_gravity()[2])) modes.append("recover" if sim.recovery else mode) tick += 1 def settle(min_steps=25, max_steps=400): """Hand back to the locomotion policy, and if the move ended on the floor let the get-up policy finish before the next move starts.""" for n in range(max_steps): do_step("walk", (0.0, 0.0, 0.0)) if n >= min_steps and sim.recovery is None: return def scaled(cmd): """Turn the unit command into this variant's real velocity limits.""" fwd, back, ang = sim.vel_limits() vx = cmd[0] * (fwd if cmd[0] >= 0 else -back) return (vx, cmd[1], cmd[2] * ang) # No eager build: a skate-only routine never pays for the legs model. for step in plan_steps: name = step["move"] variant, mode, unit_cmd = MOVES[name] switched = use(variant) if switched and log: notes.append("switched to " + variant + " at " + name + " (robot respawns at the origin)") cmd = scaled(unit_cmd) t0 = tick * D.CTRL_DT x0, y0 = float(sim.data.qpos[0]), float(sim.data.qpos[1]) # Head gestures persist until changed, like the runtime's offsets. if name in HEAD_POSES: sim.head_target[:] = np.array(HEAD_POSES[name], dtype=np.float32) * D.HEAD_MAX elif name not in ONE_SHOT and name != "sit": sim.head_target[:] = 0.0 if name in ("kick_left", "kick_right"): # Plant first: the ball is placed relative to the duck's stance, # so a kick entered mid-stride would put it under a moving foot. for _ in range(25): do_step("walk", (0.0, 0.0, 0.0)) sim.spawn_ball(rng, foot="L" if name == "kick_left" else "R") for _ in range(D.KICK_STEPS): do_step(mode, cmd) sim.post_kick_lock = D.POST_KICK_LOCK_STEPS for _ in range(D.POST_KICK_LOCK_STEPS): do_step("walk", (0.0, 0.0, 0.0)) elif name == "sit": # game.js: hold the stand under sitstand for 0.8 s, command the # sit, then give it 2.0 s to stand back up before moving on. for _ in range(int(D.SIT_HANDOVER_S / D.CTRL_DT)): do_step("sitstand", cmd, sit_flag=0.0) for _ in range(int(round(step["seconds"] / D.CTRL_DT))): do_step("sitstand", cmd, sit_flag=1.0) for _ in range(int(D.SIT_STANDUP_S / D.CTRL_DT)): do_step("sitstand", cmd, sit_flag=0.0) sim.last_action[:] = 0 elif name == "roll": tipped, n = False, 0 while n < 150: do_step(mode, cmd) n += 1 gz = float(sim.proj_gravity()[2]) if gz > -0.3: tipped = True if tipped and gz < -0.85 and n >= 40: break sim.last_action[:] = 0 settle() elif name in ("pick", "skate_crouch"): period = (D.GROUND_PICK_PERIOD_S if name == "pick" else D.CROUCH_PERIOD_S) end = (D.GROUND_PICK_END_PHASE if name == "pick" else D.CROUCH_END_PHASE) phase, n = 0.0, 0 while phase < end and n < 400: do_step(mode, cmd, phase=phase) phase += D.CTRL_DT / period n += 1 settle() else: for _ in range(int(round(step["seconds"] / D.CTRL_DT))): do_step(mode, cmd) log.append({ "move": name, "t_start": round(t0, 2), "t_end": round(tick * D.CTRL_DT, 2), "travelled_m": round(float(np.hypot(sim.data.qpos[0] - x0, sim.data.qpos[1] - y0)), 3), "variant": variant, }) telemetry = { "track": track, "upright": upright, "modes": modes, "log": log, "notes": notes, "duration_s": round(tick * D.CTRL_DT, 2), "fps": fps, "falls": sum(1 for a, b in zip(["walk"] + modes, modes) if b == "recover" and a != "recover"), "distance_m": round(float(np.hypot(sim.data.qpos[0], sim.data.qpos[1])), 3), "final_upright": round(float(sim.proj_gravity()[2]), 3), "variants": sorted({e["variant"] for e in log}), } return frames, telemetry